@edgehero/pi-dispatch 3.1.0 → 4.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.mjs CHANGED
@@ -16,6 +16,8 @@ import { effectiveCostCapMicros } from "./money.mjs";
16
16
  import { dollarWindowCaps } from "./dollar-budget.mjs";
17
17
  import { concurrencyFor, dollarCapsFor, makeInFlight, modelDollarRows, projectDollarCapsFor, projectRowFor, rowScopeFor, scopedLedgers } from "./scoped-limits.mjs";
18
18
  import { memberScopeOf, projectOf } from "./projects.mjs";
19
+ import { resolveJobSize } from "./job-size.mjs";
20
+ import { BUDGET_RECHECK_MS, HOST_BUDGET_TICK_MS, NEVER_FITS_RECHECK_MS, makeHostBudget } from "./host-budget.mjs";
19
21
  import { governedDollars } from "./allocation.mjs";
20
22
  import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
21
23
  import { makeWaitState } from "./wait-state.mjs";
@@ -90,7 +92,7 @@ const ABORT_GRACE_MS = 30_000;
90
92
  * container produces, so nothing downstream needs to know the difference -- the processor's abort
91
93
  * classification, the run record and the refund all behave exactly as they do for a stop that worked.
92
94
  */
93
- function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
95
+ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS, onStopDidNotTake = () => {}) {
94
96
  if (!signal) return run;
95
97
  return new Promise((resolve, reject) => {
96
98
  let timer = null;
@@ -106,6 +108,8 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
106
108
  if (settled) return;
107
109
  settled = true;
108
110
  log("stop_did_not_take", { job: job.id, graceMs });
111
+ // Issue #596, phase 2: the container may still run, so its host budget hold must outlive this job.
112
+ onStopDidNotTake();
109
113
  resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null, exitReason: null });
110
114
  }, graceMs);
111
115
  // A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
@@ -186,7 +190,7 @@ export function effectiveJobOf(data, settings, allowedModels = null, log = () =>
186
190
  };
187
191
  }
188
192
 
189
- export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
193
+ export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, hostBudget = null, abortGraceMs = ABORT_GRACE_MS, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", multiHost = false, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
190
194
  // The lost-lock gate's rejection lines, said ONCE per job id, stall count and attempt. The gate runs before every
191
195
  // deferral (pause window, wait, scope or endpoint busy), and BullMQ never resets a job's stall count, so a stalled
192
196
  // scheduled job whose record is refused meets the gate again on every deferred pickup: a 30-minute scope-busy hold
@@ -199,7 +203,12 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
199
203
  if (rejectedSeen.size > REJECTED_SEEN_MAX) rejectedSeen.delete(rejectedSeen.values().next().value);
200
204
  return true;
201
205
  };
202
- return async function processor(job, token, signal) {
206
+ // THE HOST BUDGET'S WAITER BOOKKEEPING (issue #596, phase 2), in ONE place around the whole pickup rather than at each
207
+ // exit: a job the budget deferred keeps its waiter (and so its hold); a job another gate deferred (a pause window, a
208
+ // wait, a full scope, a busy endpoint) keeps its waiter SUSPENDED, so it holds no room it could not use; a job that
209
+ // ended any other way (it ran, it was refused, it failed) is no waiter at all. Listing this per exit is how one of
210
+ // the many deferral sites would come to leave a live hold behind.
211
+ const pickup = async (job, token, signal, budgetState) => {
203
212
  // THE LOST-LOCK GATE, first because it is free and because it can only ever stop a run (CONST-RETRY-INFRA-ONLY).
204
213
  // BullMQ hands a job to the processor again after its stall check took it back. That happens to a job whose
205
214
  // processor FINISHED when Valkey was unreachable for longer than the lock renewal window: the record was
@@ -240,6 +249,68 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
240
249
  throw new DelayedError();
241
250
  }
242
251
 
252
+ // The limits snapshot, the project and the size, read here, right after the pause gate and ABOVE the wait gate, so
253
+ // the never-fits check below can refuse a size before a job waits (issue #596, gate round 1 of phase 2).
254
+ const limits = scopedLimits();
255
+ // The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT), resolved ONCE here from one read of the projects ref,
256
+ // beside the limits snapshot and for its reason: the gate, the ledger and the record agree for this attempt,
257
+ // whatever an operator does to projects.json mid-run. A retry or a deferral is a new pickup and resolves again.
258
+ // Every record from the never-fits check on carries it (the refusal there by hand, every one past the wait gate
259
+ // through `recordAfterGate`); the wait gate's refusals carry none and are resolved from the live ref (start.mjs).
260
+ // An id or null, never a name.
261
+ const pickupProjects = projects();
262
+ const project = projectOf(job.data, pickupProjects);
263
+ // THE JOB'S SIZE (issue #596, `job-size.mjs`), resolved ONCE here from the same limits snapshot and pickup project
264
+ // every gate below reads: the project row's memory and CPUs, else the deployment's PI_JOB_MEMORY and PI_JOB_CPUS
265
+ // (`jobSizeEnv`, validated at boot), else 4g and 2. It reaches the container as an ARGUMENT (runJob's `jobSize`,
266
+ // then `runContainer`'s `size`), never through `job.data`, so nothing queued can choose its own size, and every
267
+ // record below carries it beside the project.
268
+ const size = resolveJobSize({ project, limits, env: jobSizeEnv });
269
+
270
+ // THE NEVER-FITS CHECK (issue #596, phase 2, DES-HOST-BUDGET), right after the size and ABOVE the wait gate (gate
271
+ // round 1): a size that can never start here is known now, and a job must not hold for a day on a wait and
272
+ // THEN be told so (the wait gate's own determinate-refusals-then-holds rule). Nothing is held yet, so nothing is
273
+ // given back.
274
+ //
275
+ // A job that can run on NO OTHER HOST is refused: a size larger than this host's budget, or than its project's
276
+ // `hostShare` of it, is a determinate POLICY refusal, RETURNED before anything is spent (CONST-BUDGET-BEFORE-TOKENS,
277
+ // CONST-RETRY-INFRA-ONLY): `job-size-exceeds-host` or `job-size-exceeds-share`. The log line and the record name
278
+ // both sizes; the forge comment names neither (an issue author can act on neither). Two such jobs: one on THIS
279
+ // HOST'S OWN QUEUE (`pi-jobs@<name>`), and EVERY job on a worker with no host queue (`multiHost` false, no
280
+ // `PI_WORKER_NAME`): there the shared queue is this host's alone in all but name, since no other host that
281
+ // declared a fleet drains it, and deferring a never-fits job there (gate round 2 of phase 2) re-asked it
282
+ // every 60 s forever with no record, which is the silent no-op this project refuses.
283
+ //
284
+ // A job on the SHARED queue of a MULTI-HOST worker is NEVER refused for its size, local or forge: another
285
+ // host draining the queue may have a larger budget, or give the project a larger share of it. It is deferred for
286
+ // `NEVER_FITS_RECHECK_MS` with a named line carrying both sizes, and doctor names a project that fits no live
287
+ // host. There used to be a fleet refusal after two registry reads agreed that no live host fits, and the registry
288
+ // cannot carry that verdict: a host's row is deleted while it restarts (a clean stop) or expires after a crash, and a
289
+ // read whose HGETALL times out drops a row, so a job a restarting host would have run was refused for good.
290
+ if (hostBudget) {
291
+ await hostBudget.ready;
292
+ const misfit = hostBudget.neverFits(size, project, limits);
293
+ if (misfit !== null) {
294
+ const budgetNow = hostBudget.current();
295
+ const share = hostBudget.shareOf(project, limits);
296
+ const sizeFields = { memMiB: size.memMiB, cpuCenti: size.cpuCenti, budgetMemMiB: budgetNow.memMiB, budgetCpuCenti: budgetNow.cpuCenti, hostShare: share };
297
+ if (!multiHost || (job.queueName ?? QUEUE) !== QUEUE) {
298
+ const reason = `job-size-exceeds-${misfit}`;
299
+ deps?.log?.(reason.replaceAll("-", "_"), { jobId: job.id, project, ...sizeFields });
300
+ if (deps?.comment) await Promise.resolve(deps.comment(job.data, SIZE_REFUSAL_COMMENTS[reason])).catch(() => {});
301
+ const at = new Date(now()).toISOString();
302
+ const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false, hostBudget: { memMiB: budgetNow.memMiB, cpuCenti: budgetNow.cpuCenti, hostShare: share } };
303
+ // Above the wait gate, so through the recorder's own arguments rather than the bound one below it: the
304
+ // same pickup project and size, by hand once.
305
+ recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString(), project, size });
306
+ return result;
307
+ }
308
+ deps?.log?.("job_size_never_fits_here_deferred", { jobId: job.id, project, misfit, delayMs: NEVER_FITS_RECHECK_MS, ...sizeFields });
309
+ await job.moveToDelayed(nowMs + NEVER_FITS_RECHECK_MS, token);
310
+ throw new DelayedError();
311
+ }
312
+ }
313
+
243
314
  // The wait gate (issue #230, REQ-WAIT-FOR). THIRD: after the pause gate, because a paused job must
244
315
  // not burn a wait evaluation any more than it burns a scope re-check, and BEFORE the scope acquire,
245
316
  // because a job that is going to sit until tomorrow morning must not hold the folder mutex while it
@@ -583,7 +654,62 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
583
654
  // Before the scope acquire, so a job that cannot run on this machine at all never takes a folder
584
655
  // mutex it would immediately have to give back, and so the two releases nest rather than interleave.
585
656
  let hostHeld = false;
586
- if (hostBound) {
657
+ // THE ONE RELEASE (issue #596, phase 2). Every hold this pickup takes (the host slot, the repo or folder slot and
658
+ // the project slot with their fleet claims, the endpoint slots, and the host budget's hold) is given back here and
659
+ // nowhere else, at every exit: each gate's deferral, the setup guard and the finally. A release site that lists
660
+ // holds by hand is the one that forgets the hold added after it was written, and a bolt test refuses a bare
661
+ // release anywhere else in this function. Last taken, first given back: the endpoint holds (issue #503), then the
662
+ // scope holds (project, then repo), then the host slot, then the budget.
663
+ //
664
+ // `orphan` is the finally's: when the container's stop did not take (`stop_did_not_take`), the container may still
665
+ // run, so the budget hold becomes an ORPHAN that keeps its room until the runtime says the container is gone
666
+ // (`host-budget.mjs` `sweep`; the boot reaper is the backstop). The count slots OUTSIDE the budget (the scope, the
667
+ // project and the endpoint slots, and the host slot when there is no budget) still go back: they bound starts, and
668
+ // the 30-minute bound already ended this job. The `PI_CONCURRENCY` slot does NOT: with a budget the count is the
669
+ // budget's third dimension, so an orphan, like a seeded survivor, holds one job slot until its container
670
+ // is gone (gate round 2 of phase 2).
671
+ //
672
+ // The in-process halves go back synchronously, before the first await, so a caller that cannot await (the setup
673
+ // guard) still frees every local slot before it rethrows; the fleet halves are release-if-mine and awaited where
674
+ // the caller can. Draining, so a second call releases nothing: the in-process map's release is not idempotent.
675
+ let budgetHeld = false;
676
+ let stopDidNotTake = false;
677
+ let name;
678
+ let venue;
679
+ // THE SCOPE HOLDS, in acquire order (issue #499 part B): the repo or folder slot, then the project slot. Each is
680
+ // `{ key, fleet }`: the in-process slot under `key` (the row scope) and its fleet claim, or null.
681
+ const scopeHolds = [];
682
+ const releaseScopeHolds = () => {
683
+ const taken = scopeHolds.splice(0).reverse();
684
+ for (const hold of taken) inFlight.release(hold.key);
685
+ return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
686
+ };
687
+ // THE ENDPOINT HOLDS (issue #503), `{ id, fleet }` each, in id order.
688
+ const endpointHolds = [];
689
+ const releaseEndpointHolds = () => {
690
+ const taken = endpointHolds.splice(0).reverse();
691
+ for (const hold of taken) endpointSlots.release(hold.id);
692
+ return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
693
+ };
694
+ const releaseAllHolds = ({ orphan = false } = {}) => {
695
+ const endpointsReleased = releaseEndpointHolds();
696
+ const scopesReleased = releaseScopeHolds();
697
+ if (hostHeld) {
698
+ hostBound.slots.release(HOST_SLOT_KEY);
699
+ hostHeld = false;
700
+ }
701
+ if (budgetHeld) {
702
+ budgetHeld = false;
703
+ if (orphan) hostBudget.orphan(job.id, { name, venue, ticket: budgetState.ticket });
704
+ else hostBudget.release(job.id, { ticket: budgetState.ticket });
705
+ }
706
+ return Promise.all([endpointsReleased, scopesReleased]);
707
+ };
708
+ // With a host budget the count is the budget's own third dimension (issue #596, gate round 1 of phase 2),
709
+ // judged LAST with the memory and the CPU, so a waiting job's hold keeps a job slot too. A host slot taken here, first,
710
+ // deferred a big shared-queue job at the slot while every small that ended was replaced at once from the host queue,
711
+ // and a job deferred here never reached the budget, so it held nothing and never ran while the flood lasted.
712
+ if (hostBound && !hostBudget) {
587
713
  if (!hostBound.slots.tryAcquire(HOST_SLOT_KEY, hostBound.limit())) {
588
714
  deps?.log?.("host_busy_deferred", { jobId: job.id, delayMs: SCOPE_BUSY_RECHECK_MS });
589
715
  await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
@@ -592,46 +718,22 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
592
718
  hostHeld = true;
593
719
  }
594
720
 
595
- const limits = scopedLimits();
596
- // The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT), resolved ONCE here from one read of the projects ref,
597
- // beside the limits snapshot and for its reason: the gate, the ledger and the record agree for this attempt,
598
- // whatever an operator does to projects.json mid-run. A retry or a deferral is a new pickup and resolves again.
599
- // Every record below this line carries it (through `recordAfterGate`); a record written before this gate carries
600
- // none and is resolved from the live ref (start.mjs). An id or null, never a name.
601
- const pickupProjects = projects();
602
- const project = projectOf(job.data, pickupProjects);
603
721
  // THE ONE RECORDER BELOW THE GATE, bound once, so the pickup project is a property of the path and not of each call
604
722
  // site: every record from here on goes through it, and none can drop the field and fall back to the live ref in
605
723
  // start.mjs, which would disagree with the pickup value exactly when projects.json was edited mid-run. A bolt in
606
724
  // project-pickup.test.mjs refuses a bare `recordRun(` call below this line.
607
- const recordAfterGate = (args) => recordRun({ ...args, project });
725
+ const recordAfterGate = (args) => recordRun({ ...args, project, size });
726
+
608
727
  // The MATCHED ROW's scope keys both the in-process slot and the fleet lease (issue #498), the same string
609
728
  // `budgetCapsFor` hashes below and the boot sweeper hashes from the file: a qualified `github:acme/web` row holds
610
729
  // GitHub jobs only, a bare `acme/web` row holds every forge's under the key it always had. With no row it is the
611
730
  // job's canonical scope, so the folder mutex is keyed exactly as before.
612
731
  const scope = rowScopeFor(job.data, limits);
613
- // THE SCOPE HOLDS, in acquire order (issue #499 part B): the repo or folder slot, then the project slot. Each is
614
- // `{ key, fleet }`: the in-process slot under `key` (the row scope) and its fleet claim, or null. ONE drain gives
615
- // every hold back, last first, at every exit (a deferral, the setup guard, the finally), the endpoint holds' shape:
616
- // a release site that lists slots by hand is the one that forgets the slot added after it was written.
617
- const scopeHolds = [];
618
- // Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent. The
619
- // in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees every local
620
- // slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
621
- const releaseScopeHolds = () => {
622
- const taken = scopeHolds.splice(0).reverse();
623
- for (const hold of taken) inFlight.release(hold.key);
624
- return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
625
- };
626
732
  // Give every hold back (scope and host), then defer. Every scope-gate deferral goes through here.
627
733
  const deferScope = async (fields) => {
628
- await releaseScopeHolds();
629
- // The host slot goes back before we defer: `makeInFlight().release` is not idempotent, so a slot
630
- // held across a deferral would be a slot this machine never gets back.
631
- if (hostHeld) {
632
- hostBound.slots.release(HOST_SLOT_KEY);
633
- hostHeld = false;
634
- }
734
+ // Every hold goes back before we defer, the host slot too: `makeInFlight().release` is not idempotent, so a
735
+ // slot held across a deferral would be a slot this machine never gets back.
736
+ await releaseAllHolds();
635
737
  deps?.log?.(fields.event, { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS, ...fields.extra });
636
738
  await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
637
739
  throw new DelayedError();
@@ -729,16 +831,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
729
831
  // One snapshot per pickup (the endpoints, the overlay models, the derived set), handed to runJob as
730
832
  // `modelEndpoints` so a later gate reads the same declaration this one leased against. With no endpoints
731
833
  // declared the overlay is not even read, and the job touches nothing new: no command, no key, no field.
732
- const endpointHolds = [];
733
834
  let endpointSnapshot = null;
734
- // Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent.
735
- // The in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees
736
- // every local slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
737
- const releaseEndpointHolds = () => {
738
- const taken = endpointHolds.splice(0).reverse();
739
- for (const hold of taken) endpointSlots.release(hold.id);
740
- return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
741
- };
742
835
  if (modelEndpoints && !settingsThrew && !settings?.invalid) {
743
836
  let endpoints = [];
744
837
  let models = null;
@@ -790,19 +883,15 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
790
883
  // fleet half; the in-process bound above is exact for a server only this host reaches.
791
884
  fleet = await endpointLease.acquire(job.id, { slots: endpoint.slots, keyArgs: [hash16(endpoint.id)] });
792
885
  if (!fleet) {
793
- endpointSlots.release(endpoint.id);
886
+ // Taken in process above, so it goes back through the one release with every other hold.
887
+ endpointHolds.push({ id: endpoint.id, fleet: null });
794
888
  where = "fleet";
795
889
  } else if (fleet.degraded) {
796
890
  deps?.log?.("endpoint_lease_degraded", { jobId: job.id, endpoint: endpoint.id });
797
891
  }
798
892
  }
799
893
  if (where !== null) {
800
- await releaseEndpointHolds();
801
- await releaseScopeHolds();
802
- if (hostHeld) {
803
- hostBound.slots.release(HOST_SLOT_KEY);
804
- hostHeld = false;
805
- }
894
+ await releaseAllHolds();
806
895
  deps?.log?.("endpoint_busy_deferred", { jobId: job.id, endpoint: endpoint.id, where, delayMs: ENDPOINT_BUSY_RECHECK_MS });
807
896
  await job.moveToDelayed(nowMs + ENDPOINT_BUSY_RECHECK_MS, token);
808
897
  throw new DelayedError();
@@ -811,8 +900,40 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
811
900
  }
812
901
  }
813
902
 
903
+ // THE HOST BUDGET GATE (issue #596, phase 2, DES-HOST-BUDGET), LAST of the gates, so a job waits on the budget only
904
+ // when the budget is its only obstacle: a job a scope, its project's `concurrent` or an endpoint deferred never got
905
+ // here, so its hold (if it had one) is suspended rather than kept (`processor` above). Synchronous: the ledger is
906
+ // read, decided on and written with no await between, so two pickups on this host never both take the same room.
907
+ // A deferral, never a refusal: a full host is transient state (CONST-RETRY-INFRA-ONLY), and it is free. Two whys make
908
+ // no waiter: `running-here` (the ledger already holds this job id: another pickup of it still runs here, or its
909
+ // container outlived it as an orphan) and `unseeded` (the job containers left from before this worker started are
910
+ // not listed yet, so nothing is admitted).
911
+ // Skipped when the settings are unreadable or invalid: that job is refused or retried below without a container.
912
+ //
913
+ // The gate is told the job's VENUE (the name it names, null for the default) and the container NAME this pickup
914
+ // will use (gate round 2 of phase 2): an unread boot listing blocks only its own venue's jobs, and a
915
+ // seeded survivor or orphan whose container carries this name defers the job `running-here`, since its
916
+ // `docker run` would create a container of that very name, which the sweep would take for the survivor. A name the
917
+ // venue cannot build is null here; the registry's refusal below records that job.
918
+ if (hostBudget && !settingsThrew && !settings?.invalid) {
919
+ let budgetName = null;
920
+ try {
921
+ budgetName = containerName({ ...job.data, id: job.id });
922
+ } catch {
923
+ budgetName = null;
924
+ }
925
+ const verdict = hostBudget.gate({ id: job.id, ticket: budgetState.ticket, project, size, venue: job.data?.backend ?? null, name: typeof budgetName === "string" ? budgetName : null, getState: typeof job.getState === "function" ? () => job.getState() : null, limits });
926
+ if (!verdict.admitted) {
927
+ await releaseAllHolds();
928
+ budgetState.budgetDeferred = true;
929
+ deps?.log?.("host_budget_deferred", { jobId: job.id, project, why: verdict.why, rank: verdict.rank, memMiB: size.memMiB, cpuCenti: size.cpuCenti, delayMs: BUDGET_RECHECK_MS });
930
+ await job.moveToDelayed(nowMs + BUDGET_RECHECK_MS, token);
931
+ throw new DelayedError();
932
+ }
933
+ budgetHeld = true;
934
+ }
935
+
814
936
  let startedAt;
815
- let name;
816
937
  let timer;
817
938
  let cancelPoll;
818
939
  let cancelPolling = false;
@@ -854,7 +975,6 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
854
975
  }
855
976
  };
856
977
  let onAbort;
857
- let venue;
858
978
  try {
859
979
  // Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
860
980
  // finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
@@ -968,12 +1088,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
968
1088
  } catch (error) {
969
1089
  // Release and DRAIN: this throw never reaches the main finally below, but a shared scope must never be
970
1090
  // releasable twice -- a double release frees another holder's slot. Last taken, first given back.
971
- void releaseEndpointHolds();
972
- void releaseScopeHolds();
973
- if (hostHeld) {
974
- hostBound.slots.release(HOST_SLOT_KEY);
975
- hostHeld = false;
976
- }
1091
+ void releaseAllHolds();
977
1092
  clearTimeout(timer);
978
1093
  clearInterval(cancelPoll);
979
1094
  throw error;
@@ -1062,6 +1177,10 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
1062
1177
  // snapshot, and refuses before any reserve when it is another project's.
1063
1178
  pickupProject: project,
1064
1179
  folderProject: (folder) => projectOf({ kind: "local", folder }, pickupProjects),
1180
+ // Issue #596: the size resolved at pickup above, for the retained run's manifest and the container's argv.
1181
+ jobSize: size,
1182
+ // Issue #596, phase 2: the host's CPU budget, every job's `--cpus` while it is a number (off or unknown: null).
1183
+ ...(hostBudget && Number.isSafeInteger(hostBudget.current().cpuCenti) ? { cpuBudgetCenti: hostBudget.current().cpuCenti } : {}),
1065
1184
  // Issue #504 part B: `_other`'s ledger, the deployment cap's source and the envelope verdict, under an envelope;
1066
1185
  // absent without one, so the processor's defaults keep such a deployment byte-identical.
1067
1186
  ...(dollarInputs.otherDollars ? { otherDollars: dollarInputs.otherDollars } : {}),
@@ -1114,7 +1233,9 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
1114
1233
  // promised: the run is recorded as that cancel (the existing path, "partial work may exist") even when the
1115
1234
  // container happened to exit on its own, because the operator was told the record would say operator-cancel.
1116
1235
  runContainer: (ctx) =>
1117
- boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})).then(async (r) => {
1236
+ boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {}), abortGraceMs, () => {
1237
+ stopDidNotTake = true;
1238
+ }).then(async (r) => {
1118
1239
  await stopCancelPoll();
1119
1240
  const raced = r && r.aborted !== true && signal.aborted === true && signal.reason === "operator-cancel";
1120
1241
  if (raced) deps.log?.("cancel_acked_as_container_exited", { jobId: job.id, exitCode: r.code ?? null });
@@ -1226,7 +1347,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
1226
1347
  const endCancelled = async ({ retryable }) => {
1227
1348
  const spent = error?.budgetReserved === true;
1228
1349
  const beforeStart = retryable && !spent;
1229
- const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null), ...(error?.dollars ? { dollars: error.dollars } : {}) };
1350
+ const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null), ...(error?.dollars ? { dollars: error.dollars } : {}), ...(spent && error?.resources ? { resources: error.resources } : {}) };
1230
1351
  deps?.log?.("job_cancelled_instead_of_retry", { jobId: job.id, spent, retryable, ...(retryable ? {} : { failure: scrubCredentials(String(error?.message ?? error)).slice(0, 300) }) });
1231
1352
  if (deps?.comment) await Promise.resolve(deps.comment(job.data, beforeStart ? CANCELLED_BEFORE_START_COMMENT : TERMINAL_COMMENTS["operator-cancel"])).catch(() => {});
1232
1353
  recordAfterGate({ job, result, startedAt, endedAt: new Date().toISOString() });
@@ -1323,16 +1444,32 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
1323
1444
  // full re-check interval while the slot it wanted went free behind it. The finally is already inside
1324
1445
  // an async function, and `release` never throws. Last taken, first given back: the endpoint holds
1325
1446
  // (issue #503), then the scope holds (project, then repo), then the host slot.
1326
- const endpointsReleased = releaseEndpointHolds();
1327
- const scopesReleased = releaseScopeHolds();
1328
- if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
1329
- await endpointsReleased;
1330
- await scopesReleased;
1447
+ await releaseAllHolds({ orphan: stopDidNotTake });
1331
1448
  clearTimeout(timer);
1332
1449
  clearInterval(cancelPoll);
1333
1450
  signal.removeEventListener("abort", onAbort);
1334
1451
  }
1335
1452
  };
1453
+ return async function processor(job, token, signal) {
1454
+ if (!hostBudget) return pickup(job, token, signal, { budgetDeferred: false });
1455
+ // The pickup's TICKET: the ledger entry this pickup takes carries it, and only this pickup's release or
1456
+ // orphan can act on that entry, so a second pickup of the same job id (a stalled scheduled job handed back while
1457
+ // its first attempt still runs here) can never give back the first's hold.
1458
+ const budgetState = { budgetDeferred: false, ticket: hostBudget.enter(job.id) };
1459
+ try {
1460
+ const result = await pickup(job, token, signal, budgetState);
1461
+ hostBudget.forget(job.id);
1462
+ return result;
1463
+ } catch (error) {
1464
+ if (!budgetState.budgetDeferred) {
1465
+ if (error instanceof DelayedError) hostBudget.suspend(job.id);
1466
+ else hostBudget.forget(job.id);
1467
+ }
1468
+ throw error;
1469
+ } finally {
1470
+ hostBudget.leave(job.id);
1471
+ }
1472
+ };
1336
1473
  }
1337
1474
 
1338
1475
  /**
@@ -1341,6 +1478,17 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
1341
1478
  */
1342
1479
  export const CANCEL_STOP_BOUND_MS = 1_000;
1343
1480
 
1481
+ /**
1482
+ * The forge comments for the two never-fits refusals (issue #596, phase 2). GENERIC on purpose: never the size, the
1483
+ * budget or the project, which are operator configuration an issue author can act on none of. The worker log, the run
1484
+ * record and doctor name both sizes. There is no fleet refusal (gate round 1 of phase 2): a job on the shared
1485
+ * queue waits for a host it fits on.
1486
+ */
1487
+ export const SIZE_REFUSAL_COMMENTS = Object.freeze({
1488
+ "job-size-exceeds-host": "Refused: this job's size is larger than the worker host's job budget, so it could never start there. No container was started and nothing was spent. Ask the operator to lower this project's job size or raise the host budget. Not run.",
1489
+ "job-size-exceeds-share": "Refused: this job's size is larger than the share of the worker host's job budget its project may use, so it could never start there. No container was started and nothing was spent. Ask the operator to lower this project's job size or raise its host share. Not run.",
1490
+ });
1491
+
1344
1492
  /** The comment for a job the operator cancelled before it started, where it would have been held or retried (gate of PR #479). */
1345
1493
  export const CANCELLED_BEFORE_START_COMMENT = "Stopped: the operator cancelled this run before it started. Nothing was spent. Not retried.";
1346
1494
 
@@ -1369,7 +1517,7 @@ export function keeperHoldState(job, at) {
1369
1517
  return { at, since, startedMs };
1370
1518
  }
1371
1519
 
1372
- export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [] }) {
1520
+ export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [], hostBudget: hostBudgetOptions = null }) {
1373
1521
  // One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
1374
1522
  // check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
1375
1523
  // because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
@@ -1395,6 +1543,20 @@ export function createWorker({ connection, name, stopContainer, containerName, h
1395
1543
  const hostBound = hostQueue ? { slots: hostSlots, limit: () => liveConcurrency() } : null;
1396
1544
  const liveConcurrency = () => workers[0]?.concurrency ?? concurrency;
1397
1545
 
1546
+ // THE HOST BUDGET (issue #596, phase 2, DES-HOST-BUDGET), built ONCE here and handed to every processor, for the host
1547
+ // slot's reason: it bounds the MACHINE, so two queues with two ledgers would each admit a full budget. Unlike the host
1548
+ // slot it is armed with ONE queue too, because a single Worker's count still cannot tell a 20g job from a 2g one.
1549
+ // `hostBudgetOptions` is start.mjs's (the settings, the default size, the facts reader, the orphan check); a bare
1550
+ // wiring passes none and keeps today's behaviour. The tick re-reads the facts, verifies stale holds and sweeps
1551
+ // orphans, off every job path, unref'd and cleared on stop like the registry's beat.
1552
+ // The live `PI_CONCURRENCY` is the budget's third dimension, so with a budget the host slot above is not taken.
1553
+ const hostBudget = hostBudgetOptions ? makeHostBudget({ scopedLimits, countLimit: () => liveConcurrency(), ...hostBudgetOptions }) : null;
1554
+ let budgetTick = null;
1555
+ if (hostBudget) {
1556
+ budgetTick = setInterval(() => void hostBudget.tick(), hostBudgetOptions.tickMs ?? HOST_BUDGET_TICK_MS);
1557
+ budgetTick.unref?.();
1558
+ }
1559
+
1398
1560
  for (const queueName of names) {
1399
1561
  let worker; // referenced by cancelJob/applyConcurrency before assignment; only called later, so the TDZ is fine
1400
1562
  const processor = makeProcessor({
@@ -1413,6 +1575,11 @@ export function createWorker({ connection, name, stopContainer, containerName, h
1413
1575
  containerName,
1414
1576
  redis,
1415
1577
  getSettings,
1578
+ // Issue #596, phase 2: the ONE host budget. `multiHost` is whether this worker declared a fleet (a host queue):
1579
+ // without one it declares no fleet and no other host is there to wait for, so a size that never fits here is refused rather than
1580
+ // deferred for a host that does not exist (gate round 2 of phase 2).
1581
+ hostBudget,
1582
+ multiHost: hostQueue !== null,
1416
1583
  // Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
1417
1584
  // next job start, and with two queues both have to move or the host bound and the queue bounds
1418
1585
  // stop agreeing. Guarded so only an integer that actually differs touches the property.
@@ -1426,6 +1593,8 @@ export function createWorker({ connection, name, stopContainer, containerName, h
1426
1593
  // independent maps would double every one of them exactly as two Workers double concurrency.
1427
1594
  scopedLimits,
1428
1595
  projects,
1596
+ // Issue #596: the deployment's default job size settings, read beside the limits snapshot at every pickup.
1597
+ jobSizeEnv,
1429
1598
  allocation,
1430
1599
  inFlight,
1431
1600
  hostBound,
@@ -1481,12 +1650,15 @@ export function createWorker({ connection, name, stopContainer, containerName, h
1481
1650
  // The host-queue worker, for the caller that must register listeners on both. Attached rather than
1482
1651
  // returned as a pair so every existing caller keeps receiving exactly what it received before.
1483
1652
  primary.hostWorker = workers[1] ?? null;
1653
+ // Issue #596, phase 2: the host budget, for the registry beat's thunks (start.mjs) and doctor-facing snapshots.
1654
+ primary.hostBudget = hostBudget;
1484
1655
 
1485
1656
  const stop = async () => {
1486
1657
  // Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
1487
1658
  // worker.close() would wait up to 30 minutes for the container. ONE shutdown for every queue: two
1488
1659
  // registrations would mean two `process.exit(0)` racing, and the second worker's containers would
1489
1660
  // outlive the handler that was meant to stop them.
1661
+ if (budgetTick) clearInterval(budgetTick);
1490
1662
  for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
1491
1663
  for (const w of workers) await w.close().catch(() => {});
1492
1664
  // Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.