@edgehero/pi-dispatch 2.0.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +44 -7
- package/README.md +14 -6
- package/deploy/com.pi-dispatch.worker.plist +1 -1
- package/deploy/docker-compose.yml +12 -0
- package/deploy/egress-proxy.conf +28 -3
- package/deploy/pi-dispatch-egress-proxy.container +8 -2
- package/deploy/worker-env-wrapper.cmd +1 -1
- package/deploy/worker-env-wrapper.sh +3 -3
- package/package.json +9 -2
- package/src/allocation.mjs +731 -0
- package/src/backends.mjs +243 -0
- package/src/budget.mjs +40 -4
- package/src/cli.mjs +222 -11
- package/src/config.mjs +126 -5
- package/src/daemon-facts.mjs +3 -0
- package/src/deployment-venue.mjs +1 -0
- package/src/doctor.mjs +2316 -183
- package/src/dollar-budget.mjs +373 -0
- package/src/dollar-fingerprint.mjs +83 -0
- package/src/egress-cli.mjs +316 -0
- package/src/egress-proxy-state.mjs +35 -5
- package/src/egress.mjs +16 -3
- package/src/env-allowlist.mjs +142 -18
- package/src/env-file.mjs +194 -25
- package/src/envelope.mjs +413 -0
- package/src/exit-code.mjs +22 -0
- package/src/fleet-lease.mjs +85 -25
- package/src/get-token.mjs +16 -5
- package/src/git-dirty.mjs +67 -0
- package/src/github-app-setup.mjs +6 -3
- package/src/github-host.mjs +5 -3
- package/src/host-pi.mjs +19 -3
- package/src/identity.mjs +2 -1
- package/src/image-preflight.mjs +98 -24
- package/src/image-ref.mjs +37 -0
- package/src/import-pi.mjs +4 -2
- package/src/index.mjs +407 -62
- package/src/init.mjs +18 -0
- package/src/job-id.mjs +26 -3
- package/src/live-probes.mjs +24 -9
- package/src/model-catalog.mjs +297 -0
- package/src/model-endpoints.mjs +649 -0
- package/src/model-ref.mjs +151 -0
- package/src/models-json.mjs +262 -0
- package/src/money.mjs +144 -0
- package/src/octokit-log.mjs +65 -0
- package/src/outbox-plan.mjs +218 -0
- package/src/outbox.mjs +29 -9
- package/src/output-cap.mjs +157 -0
- package/src/packages.mjs +2 -2
- package/src/pause-windows.mjs +81 -2
- package/src/pi-model-loader.mjs +77 -0
- package/src/podman-stack.mjs +16 -3
- package/src/portfolio-snapshot.mjs +304 -0
- package/src/prepare-local.mjs +247 -12
- package/src/prepare.mjs +35 -3
- package/src/pricing.mjs +9 -5
- package/src/priorities.mjs +569 -0
- package/src/processor.mjs +603 -173
- package/src/project-id.mjs +17 -0
- package/src/projects.mjs +238 -0
- package/src/provider-key.mjs +32 -7
- package/src/provider-steering.mjs +214 -59
- package/src/queue.mjs +111 -6
- package/src/reserved-env.mjs +30 -0
- package/src/run-container.mjs +59 -5
- package/src/run-history.mjs +379 -24
- package/src/run-mirror.mjs +30 -0
- package/src/runtime-settings.mjs +104 -9
- package/src/schedules.mjs +33 -1
- package/src/scoped-limits.mjs +447 -27
- package/src/secrets.mjs +2 -1
- package/src/service.mjs +15 -4
- package/src/session-store.mjs +131 -6
- package/src/start.mjs +528 -40
- package/src/subscriptions.mjs +7 -3
- package/src/triggers-file.mjs +65 -4
- package/src/triggers.mjs +140 -9
- package/src/up.mjs +308 -34
- package/src/valkey-endpoint.mjs +3 -2
package/src/index.mjs
CHANGED
|
@@ -8,7 +8,15 @@ import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING, TERMINAL
|
|
|
8
8
|
import { NETNS_KEEPER_YOUNG_HOLD_MAX_MS, netnsKeeperCrashLoopSentence, netnsKeeperLoopAgainSentence } from "./netns-keeper.mjs";
|
|
9
9
|
import { PODMAN_RESTART_HOLD_EXPIRED, PODMAN_RESTART_HOLD_MAX_MS, PODMAN_RESTART_HOLD_RECHECK_MS } from "./runtime-observations.mjs";
|
|
10
10
|
import { targetFor } from "./run-history.mjs";
|
|
11
|
-
import {
|
|
11
|
+
import { isPerMachineHost } from "./backends.mjs";
|
|
12
|
+
import { hash16 } from "./fleet-lease.mjs";
|
|
13
|
+
import { endpointsForModel } from "./model-endpoints.mjs";
|
|
14
|
+
import { splitModelEntry } from "./model-ref.mjs";
|
|
15
|
+
import { effectiveCostCapMicros } from "./money.mjs";
|
|
16
|
+
import { dollarWindowCaps } from "./dollar-budget.mjs";
|
|
17
|
+
import { concurrencyFor, dollarCapsFor, makeInFlight, modelDollarRows, projectDollarCapsFor, projectRowFor, rowScopeFor, scopedLedgers } from "./scoped-limits.mjs";
|
|
18
|
+
import { memberScopeOf, projectOf } from "./projects.mjs";
|
|
19
|
+
import { governedDollars } from "./allocation.mjs";
|
|
12
20
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
13
21
|
import { makeWaitState } from "./wait-state.mjs";
|
|
14
22
|
|
|
@@ -18,6 +26,17 @@ export const QUEUE = "pi-jobs";
|
|
|
18
26
|
/** The key the host-wide in-flight count lives under. One machine, one counter, whatever the queue. */
|
|
19
27
|
export const HOST_SLOT_KEY = "host";
|
|
20
28
|
export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* The failed reason BullMQ gives a job its stall check failed (`maxStalledCount: 0` below): the literal in the pinned
|
|
32
|
+
* bullmq's `moveStalledJobsToWait`, stored as the job's deferred failure and thrown as an UnrecoverableError at the
|
|
33
|
+
* next pickup. Matched EXACTLY by start.mjs's failed listener, and pinned against the installed bullmq source by a
|
|
34
|
+
* test, so a bullmq bump that rewords it fails that test rather than silently turning the lost-lock check off.
|
|
35
|
+
*/
|
|
36
|
+
export const STALLED_FAILED_REASON = "job stalled more than allowable limit";
|
|
37
|
+
|
|
38
|
+
/** How many (job, stall count, attempt, source) keys the processor remembers having logged a refused record for. */
|
|
39
|
+
export const REJECTED_SEEN_MAX = 1000;
|
|
21
40
|
// The scope-busy re-check (issue #242): a held scope has no natural "until" (the holder may run to
|
|
22
41
|
// JOB_TIMEOUT_MS), so a deferred job re-tests on a fixed cadence. 5s keeps the worst case trivial
|
|
23
42
|
// (<=360 wakes across a 30-minute hold, each ~1ms of synchronous predicate briefly occupying a slot)
|
|
@@ -26,6 +45,13 @@ export const JOB_TIMEOUT_MS = 30 * 60 * 1000; // REQ-JOB-TIMEOUT-30M
|
|
|
26
45
|
// docker daemon bounds any herd by its own concurrency, and a contended wake just re-defers.
|
|
27
46
|
export const SCOPE_BUSY_RECHECK_MS = 5_000;
|
|
28
47
|
|
|
48
|
+
// The endpoint-busy re-check (issue #503): a job whose model server has every slot held re-tests on a fixed
|
|
49
|
+
// cadence, for the scope re-check's reason (a held slot has no natural "until"). Its own value rather than a
|
|
50
|
+
// borrow of 5s, because nothing records WHY a job sits in the delayed set and the wake instant is the only
|
|
51
|
+
// evidence an operator has: 7s is distinct from the scope re-check, the wait throttle floor and the supersede
|
|
52
|
+
// re-ask, and a test keeps all four apart. Short, because a local model run is often short too.
|
|
53
|
+
export const ENDPOINT_BUSY_RECHECK_MS = 7_000;
|
|
54
|
+
|
|
29
55
|
// How long a job waits before re-asking whether a target's holder is still alive (issue #230). Reached only
|
|
30
56
|
// when the liveness probe could not answer, which is a redis or queue fault rather than a normal state, so
|
|
31
57
|
// this is a short retry rather than a cadence: the job is deciding nothing and holding nothing while it
|
|
@@ -111,9 +137,93 @@ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
|
111
137
|
* under `job.data > overlay > env` precedence and re-binds the worker slot count via `applyConcurrency`.
|
|
112
138
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
113
139
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
140
|
+
*
|
|
141
|
+
* The read happens once per pickup, right after the scope gate and before the model endpoint gate (issue #503),
|
|
142
|
+
* because that gate needs the effective provider and model, which the overlay can supply.
|
|
143
|
+
*/
|
|
144
|
+
/**
|
|
145
|
+
* A job's endpoint set (issues #503, #502): the endpoints its main model AND every model on its effective allowed
|
|
146
|
+
* list are served by, so a job that may switch to a listed model on another local server holds that server's slot
|
|
147
|
+
* too. The caller deduplicates by endpoint id. With no list it is the main model's alone, and the residual stays
|
|
148
|
+
* as #503 named it: an unrestricted job's mid-run switch to an undeclared model takes no slot. The keyless verdict
|
|
149
|
+
* is a different question ("every model of the provider", `keylessVerdict`) and does not read this set.
|
|
150
|
+
*
|
|
151
|
+
* `models` here is the parsed overlay models.json (the endpoint module's name for it); the list is `job.models`.
|
|
152
|
+
*/
|
|
153
|
+
export function mainModelEndpoints({ models, job, endpoints }) {
|
|
154
|
+
const set = endpointsForModel({ models, provider: job.provider, modelId: job.model, endpoints });
|
|
155
|
+
for (const entry of Array.isArray(job.models) ? job.models : []) {
|
|
156
|
+
const ref = splitModelEntry(entry);
|
|
157
|
+
if (ref !== null) set.push(...endpointsForModel({ models, provider: ref.provider, modelId: ref.model, endpoints }));
|
|
158
|
+
}
|
|
159
|
+
return set;
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* The job runJob is handed: `job.data > overlay > env` precedence, field by field (INT-CONFIG-OVERLAY-CONTRACT).
|
|
164
|
+
* ONE function for the two readers (the endpoint gate at pickup and the runJob call), so the gate can never lease
|
|
165
|
+
* for a model other than the one the container is started with.
|
|
114
166
|
*/
|
|
115
|
-
export function
|
|
167
|
+
export function effectiveJobOf(data, settings, allowedModels = null, log = () => {}) {
|
|
168
|
+
// Issue #502: the allowed-model list is the trigger's, else the deployment's PI_ALLOWED_MODELS, else none. The
|
|
169
|
+
// env list arrives as its own argument and never through `settings`, which the overlay (and so `dispatch_set`)
|
|
170
|
+
// writes. Absent stays absent, so an unrestricted job's effective job has no `models` key at all.
|
|
171
|
+
const models = data.models ?? allowedModels ?? null;
|
|
172
|
+
return {
|
|
173
|
+
...data,
|
|
174
|
+
provider: data.provider ?? settings.provider,
|
|
175
|
+
model: data.model ?? settings.model,
|
|
176
|
+
maxTurns: data.maxTurns ?? settings.maxTurns,
|
|
177
|
+
maxTokens: data.maxTokens ?? settings.maxTokens, // optional per-job token budget (issue #25); null => runner meter only
|
|
178
|
+
...(models !== null ? { models } : {}),
|
|
179
|
+
// Issue #501: the per-job dollar cap in integer micro-dollars, or null for none. NOT `??` like the fields
|
|
180
|
+
// above: a trigger's `run.maxCostUsd` may only NARROW, so this is the smaller of the trigger's and the
|
|
181
|
+
// deployment's, each counted only when set (`effectiveCostCapMicros` says why a trigger-only cap applies
|
|
182
|
+
// and what a malformed job value reads as). buildContainerEnv sends it as PI_MAX_COST_MICROS, and the
|
|
183
|
+
// image preflight requires `costCap` of any job that carries one.
|
|
184
|
+
// A malformed queued value is logged by KEY only (`job_cost_cap_malformed`), never by value.
|
|
185
|
+
maxCostMicros: effectiveCostCapMicros(data.maxCostUsd, settings.maxCostUsd, (key) => log("job_cost_cap_malformed", { key })),
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
190
|
+
// The lost-lock gate's rejection lines, said ONCE per job id, stall count and attempt. The gate runs before every
|
|
191
|
+
// deferral (pause window, wait, scope or endpoint busy), and BullMQ never resets a job's stall count, so a stalled
|
|
192
|
+
// scheduled job whose record is refused meets the gate again on every deferred pickup: a 30-minute scope-busy hold
|
|
193
|
+
// is about 360 of them. The verdict cannot change between them (same record, same attempt), so one line says it.
|
|
194
|
+
// Bounded, oldest out first, so a long-lived worker holds at most REJECTED_SEEN_MAX keys.
|
|
195
|
+
const rejectedSeen = new Set();
|
|
196
|
+
const firstRejection = (key) => {
|
|
197
|
+
if (rejectedSeen.has(key)) return false;
|
|
198
|
+
rejectedSeen.add(key);
|
|
199
|
+
if (rejectedSeen.size > REJECTED_SEEN_MAX) rejectedSeen.delete(rejectedSeen.values().next().value);
|
|
200
|
+
return true;
|
|
201
|
+
};
|
|
116
202
|
return async function processor(job, token, signal) {
|
|
203
|
+
// THE LOST-LOCK GATE, first because it is free and because it can only ever stop a run (CONST-RETRY-INFRA-ONLY).
|
|
204
|
+
// BullMQ hands a job to the processor again after its stall check took it back. That happens to a job whose
|
|
205
|
+
// processor FINISHED when Valkey was unreachable for longer than the lock renewal window: the record was
|
|
206
|
+
// written, the completion was refused ("Missing lock"), and the job stayed active without a lock. A plain job
|
|
207
|
+
// then carries a deferred failure and never reaches here (start.mjs's failed listener handles it), but a
|
|
208
|
+
// scheduled job is moved back to wait and would run again, PAID, with its second record overwriting the
|
|
209
|
+
// first. So a job that has stalled (`stalledCounter > 0`) and whose record says this same attempt finished
|
|
210
|
+
// without failing ends as that record says, without a container. `budgetReserved` is the record's own: the
|
|
211
|
+
// completed listener then pages exactly when it would have for the first finish, which never reached it.
|
|
212
|
+
// No record (a worker that died mid-run, a lookup fault) keeps today's path: the job runs, and the stall
|
|
213
|
+
// guard bounds how often.
|
|
214
|
+
if (Number(job.stalledCounter) > 0 && typeof settledRecord === "function") {
|
|
215
|
+
const attempt = (Number.isInteger(job.attemptsMade) && job.attemptsMade >= 0 ? job.attemptsMade : 0) + 1;
|
|
216
|
+
// A record found and refused is said, with its fixed reason, because the job then RUNS again (paid).
|
|
217
|
+
const onReject = (reason, source) => {
|
|
218
|
+
if (firstRejection(`${job.id}\u0000${job.stalledCounter}\u0000${attempt}\u0000${source}`)) deps?.log?.("job_lost_lock_record_rejected", { jobId: job.id, reason, source });
|
|
219
|
+
};
|
|
220
|
+
const record = await settledRecord(job.id, { attempt, since: job.timestamp, onReject });
|
|
221
|
+
if (record) {
|
|
222
|
+
deps?.log?.("job_lost_lock_after_completion", { jobId: job.id, outcome: record.outcome, ...(record.reason ? { reason: record.reason } : {}) });
|
|
223
|
+
return { outcome: record.outcome, reason: record.reason ?? null, exitCode: record.exitCode ?? null, turns: record.turns ?? null, tokens: record.tokens ?? null, budgetReserved: record.budgetReserved ?? null };
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
|
|
117
227
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
118
228
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
119
229
|
// auto-resumes when re-picked. This is FIRST, before the kill timer, the settings read, and the budget
|
|
@@ -448,14 +558,14 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
448
558
|
}
|
|
449
559
|
|
|
450
560
|
// Per-scope concurrency and the one-job-per-folder mutex (issue #242,
|
|
451
|
-
// INT-SCOPED-LIMITS-FILE-CONTRACT).
|
|
561
|
+
// INT-SCOPED-LIMITS-FILE-CONTRACT). After the pause gate (a paused job must
|
|
452
562
|
// not burn re-check wakes) and after the wait gate (a job holding until tomorrow must not sit on a
|
|
453
563
|
// folder while it does), and STRICTLY above the `try` below, like the pause gate and for the same two
|
|
454
564
|
// reasons: a DelayedError thrown inside the try would be converted to UnrecoverableError by the
|
|
455
565
|
// catch, and a moveToDelayed rejection here must escape RAW into BullMQ's normal failed-attempt
|
|
456
566
|
// handling exactly as the pause gate's does (inside the try it would become a permanent failure
|
|
457
567
|
// plus a failure record for what was a transient blip). The limits snapshot is read ONCE here and
|
|
458
|
-
// shared with `
|
|
568
|
+
// shared with `scopedLedgers` below, so the gate and the money ledger cannot disagree mid-job.
|
|
459
569
|
// tryAcquire is a synchronous check-and-increment -- no await between read and take, so Node's
|
|
460
570
|
// single thread makes it atomic at any concurrency -- and the local-folder limit is a structural 1
|
|
461
571
|
// (concurrencyFor) with no file and no off-switch: the scheduler mints a cron trigger's next
|
|
@@ -483,26 +593,59 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
483
593
|
}
|
|
484
594
|
|
|
485
595
|
const limits = scopedLimits();
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
596
|
+
// The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT), resolved ONCE here from one read of the projects ref,
|
|
597
|
+
// beside the limits snapshot and for its reason: the gate, the ledger and the record agree for this attempt,
|
|
598
|
+
// whatever an operator does to projects.json mid-run. A retry or a deferral is a new pickup and resolves again.
|
|
599
|
+
// Every record below this line carries it (through `recordAfterGate`); a record written before this gate carries
|
|
600
|
+
// none and is resolved from the live ref (start.mjs). An id or null, never a name.
|
|
601
|
+
const pickupProjects = projects();
|
|
602
|
+
const project = projectOf(job.data, pickupProjects);
|
|
603
|
+
// THE ONE RECORDER BELOW THE GATE, bound once, so the pickup project is a property of the path and not of each call
|
|
604
|
+
// site: every record from here on goes through it, and none can drop the field and fall back to the live ref in
|
|
605
|
+
// start.mjs, which would disagree with the pickup value exactly when projects.json was edited mid-run. A bolt in
|
|
606
|
+
// project-pickup.test.mjs refuses a bare `recordRun(` call below this line.
|
|
607
|
+
const recordAfterGate = (args) => recordRun({ ...args, project });
|
|
608
|
+
// The MATCHED ROW's scope keys both the in-process slot and the fleet lease (issue #498), the same string
|
|
609
|
+
// `budgetCapsFor` hashes below and the boot sweeper hashes from the file: a qualified `github:acme/web` row holds
|
|
610
|
+
// GitHub jobs only, a bare `acme/web` row holds every forge's under the key it always had. With no row it is the
|
|
611
|
+
// job's canonical scope, so the folder mutex is keyed exactly as before.
|
|
612
|
+
const scope = rowScopeFor(job.data, limits);
|
|
613
|
+
// THE SCOPE HOLDS, in acquire order (issue #499 part B): the repo or folder slot, then the project slot. Each is
|
|
614
|
+
// `{ key, fleet }`: the in-process slot under `key` (the row scope) and its fleet claim, or null. ONE drain gives
|
|
615
|
+
// every hold back, last first, at every exit (a deferral, the setup guard, the finally), the endpoint holds' shape:
|
|
616
|
+
// a release site that lists slots by hand is the one that forgets the slot added after it was written.
|
|
617
|
+
const scopeHolds = [];
|
|
618
|
+
// Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent. The
|
|
619
|
+
// in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees every local
|
|
620
|
+
// slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
|
|
621
|
+
const releaseScopeHolds = () => {
|
|
622
|
+
const taken = scopeHolds.splice(0).reverse();
|
|
623
|
+
for (const hold of taken) inFlight.release(hold.key);
|
|
624
|
+
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
625
|
+
};
|
|
626
|
+
// Give every hold back (scope and host), then defer. Every scope-gate deferral goes through here.
|
|
627
|
+
const deferScope = async (fields) => {
|
|
628
|
+
await releaseScopeHolds();
|
|
629
|
+
// The host slot goes back before we defer: `makeInFlight().release` is not idempotent, so a slot
|
|
630
|
+
// held across a deferral would be a slot this machine never gets back.
|
|
631
|
+
if (hostHeld) {
|
|
632
|
+
hostBound.slots.release(HOST_SLOT_KEY);
|
|
633
|
+
hostHeld = false;
|
|
634
|
+
}
|
|
635
|
+
deps?.log?.(fields.event, { jobId: job.id, kind: job.data?.kind === "local" ? "local" : "forge", delayMs: SCOPE_BUSY_RECHECK_MS, ...fields.extra });
|
|
636
|
+
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
637
|
+
throw new DelayedError();
|
|
638
|
+
};
|
|
489
639
|
if (scope) {
|
|
490
640
|
const ceiling = concurrencyFor(job.data, limits);
|
|
491
641
|
if (!inFlight.tryAcquire(scope, ceiling)) {
|
|
492
642
|
// Optional-chained: makeProcessor gives `deps` no default and bare wirings pass deps: {}.
|
|
493
643
|
// The scope itself stays out of the log line (no-pii-in-logs -- a local scope is a full
|
|
494
644
|
// host path); the delayed count and the job id are what an operator needs to see it.
|
|
495
|
-
|
|
496
|
-
// The host slot goes back before we defer: `makeInFlight().release` is not idempotent, so a slot
|
|
497
|
-
// held across a deferral would be a slot this machine never gets back.
|
|
498
|
-
if (hostHeld) {
|
|
499
|
-
hostBound.slots.release(HOST_SLOT_KEY);
|
|
500
|
-
hostHeld = false;
|
|
501
|
-
}
|
|
502
|
-
await job.moveToDelayed(nowMs + SCOPE_BUSY_RECHECK_MS, token);
|
|
503
|
-
throw new DelayedError();
|
|
645
|
+
await deferScope({ event: "scope_busy_deferred" });
|
|
504
646
|
}
|
|
505
|
-
|
|
647
|
+
const repoHold = { key: scope, fleet: null };
|
|
648
|
+
scopeHolds.push(repoHold);
|
|
506
649
|
|
|
507
650
|
// THE FLEET-WIDE HALF of a scoped ceiling (issue #57). A `scoped-limits.json` row's day/week/month
|
|
508
651
|
// caps are already atomic INCRs on shared keys; its `concurrent` was a per-process Map, so it
|
|
@@ -519,20 +662,152 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
519
662
|
// And an unlimited forge scope never claims either: `concurrencyFor` returns Infinity with no
|
|
520
663
|
// matching row, so a deployment with no scoped-limits file issues no command at all.
|
|
521
664
|
if (scopeLease && job.data?.kind !== "local" && Number.isFinite(ceiling)) {
|
|
522
|
-
|
|
523
|
-
if (!
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
665
|
+
repoHold.fleet = await scopeLease.acquire(job.id, { slots: ceiling, keyArgs: [hash16(scope)] });
|
|
666
|
+
if (!repoHold.fleet) await deferScope({ event: "scope_busy_deferred", extra: { where: "fleet" } });
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
// THE PROJECT SLOT (issue #499 part B): a project row's `concurrent` bounds every member of the project together.
|
|
671
|
+
// Taken AFTER the repo slot, in the same order for every job, so two jobs cannot each hold one and wait on the
|
|
672
|
+
// other; given back with it, last first, by the one drain above. Keyed by the project ROW's scope
|
|
673
|
+
// (`project:<id>`): the in-process slot under that string, the fleet lease under `slot:s:<hash16(project:<id>)>`,
|
|
674
|
+
// which is exactly the key the boot sweeper hashes from the row. A LOCAL member takes the fleet half too, where
|
|
675
|
+
// its own folder slot does not: a folder path carries no identity across hosts, but a project id does (every
|
|
676
|
+
// host carries the same projects.json, INT-PROJECTS-FILE-CONTRACT). A deferral, never a refusal, like the repo's.
|
|
677
|
+
const projectRow = projectRowFor(limits, project);
|
|
678
|
+
if (projectRow && Number.isSafeInteger(projectRow.concurrent)) {
|
|
679
|
+
if (!inFlight.tryAcquire(projectRow.scope, projectRow.concurrent)) {
|
|
680
|
+
await deferScope({ event: "scope_busy_deferred", extra: { ledger: "project" } });
|
|
681
|
+
}
|
|
682
|
+
const projectHold = { key: projectRow.scope, fleet: null };
|
|
683
|
+
scopeHolds.push(projectHold);
|
|
684
|
+
if (scopeLease) {
|
|
685
|
+
projectHold.fleet = await scopeLease.acquire(job.id, { slots: projectRow.concurrent, keyArgs: [hash16(projectRow.scope)] });
|
|
686
|
+
if (!projectHold.fleet) await deferScope({ event: "scope_busy_deferred", extra: { ledger: "project", where: "fleet" } });
|
|
687
|
+
}
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
// THE ONE SETTINGS READ (issue #503), hoisted here from the top of the main `try` below because the
|
|
691
|
+
// endpoint gate after it needs the effective provider and model, which the overlay can supply. Still ONE
|
|
692
|
+
// call per pickup: two reads could straddle an overlay edit, and the gate would then lease for one model
|
|
693
|
+
// while the container ran another. A throw is CAPTURED, never raised here: it is re-raised at the old spot
|
|
694
|
+
// inside the main `try`, so its record, its retry decision and the releases are exactly what they were.
|
|
695
|
+
let settings = null;
|
|
696
|
+
let settingsError = null;
|
|
697
|
+
let settingsThrew = false;
|
|
698
|
+
try {
|
|
699
|
+
settings = await getSettings();
|
|
700
|
+
} catch (error) {
|
|
701
|
+
settingsThrew = true;
|
|
702
|
+
settingsError = error;
|
|
703
|
+
}
|
|
704
|
+
|
|
705
|
+
// THE MODEL ENDPOINT GATE (issue #503, INT-MODEL-ENDPOINTS-FILE-CONTRACT). A declared local model server
|
|
706
|
+
// has a fixed number of parallel slots, and nothing else bounds how many jobs pile onto it: three per host
|
|
707
|
+
// on several hosts, each metering $0. So a job whose main model, or any model on its effective allowed-model
|
|
708
|
+
// list (issue #502: `run.models`, else PI_ALLOWED_MODELS), is served by a declared endpoint takes one of each
|
|
709
|
+
// such endpoint's slots here and holds it until its container is gone (`endpointSetFor`). An UNRESTRICTED job's
|
|
710
|
+
// set is its main model's alone, so its mid-run switch to another declared model is not counted, a named residual.
|
|
711
|
+
//
|
|
712
|
+
// LOCAL JOBS TAKE IT TOO, where they skip the fleet scope lease: a folder path carries no identity across
|
|
713
|
+
// hosts, but an endpoint is one physical server whoever calls it.
|
|
714
|
+
//
|
|
715
|
+
// Two halves, like the scope's: the in-process bound (`endpointSlots`, shared by both Workers like the host
|
|
716
|
+
// slot) and the fleet lease, which exists only with a declared worker name. A Valkey fault fails the fleet
|
|
717
|
+
// half OPEN, and the in-process bound underneath is then the whole bound for this host, said in the log.
|
|
718
|
+
//
|
|
719
|
+
// In ID ORDER, so two jobs on two shared endpoints cannot each hold one and wait on the other: every job
|
|
720
|
+
// asks in the same order, which is what rules the cycle out. On any miss EVERYTHING taken so far goes back
|
|
721
|
+
// (endpoint holds, the scope's fleet claim, its in-process slot, the host slot) before the deferral: an
|
|
722
|
+
// in-process release is not idempotent, so a slot kept across a deferral is one this host never gets back,
|
|
723
|
+
// and a deferred job holding slots would starve the jobs that could run.
|
|
724
|
+
//
|
|
725
|
+
// A deferral, never a refusal: a full server is transient state (CONST-RETRY-INFRA-ONLY), and it is free,
|
|
726
|
+
// decided before any token, clone or reservation (CONST-BUDGET-BEFORE-TOKENS). Skipped when the settings
|
|
727
|
+
// are unreadable or invalid: that job is refused or retried below without starting anything.
|
|
728
|
+
//
|
|
729
|
+
// One snapshot per pickup (the endpoints, the overlay models, the derived set), handed to runJob as
|
|
730
|
+
// `modelEndpoints` so a later gate reads the same declaration this one leased against. With no endpoints
|
|
731
|
+
// declared the overlay is not even read, and the job touches nothing new: no command, no key, no field.
|
|
732
|
+
const endpointHolds = [];
|
|
733
|
+
let endpointSnapshot = null;
|
|
734
|
+
// Drains the holds, so a second call releases nothing: the in-process map's release is not idempotent.
|
|
735
|
+
// The in-process half goes back synchronously, so a caller that cannot await (the setup guard) still frees
|
|
736
|
+
// every local slot before it rethrows; the fleet half is release-if-mine and awaited where it can be.
|
|
737
|
+
const releaseEndpointHolds = () => {
|
|
738
|
+
const taken = endpointHolds.splice(0).reverse();
|
|
739
|
+
for (const hold of taken) endpointSlots.release(hold.id);
|
|
740
|
+
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
741
|
+
};
|
|
742
|
+
if (modelEndpoints && !settingsThrew && !settings?.invalid) {
|
|
743
|
+
let endpoints = [];
|
|
744
|
+
let models = null;
|
|
745
|
+
let set = [];
|
|
746
|
+
// A failure to READ the overlay (any errno `readOverlayModels` rethrows), carried in the snapshot with its code,
|
|
747
|
+
// so the credential gate gives no keyless verdict on it. The model gate runs first and reads the file too: it
|
|
748
|
+
// refuses every job on a permanent errno (EACCES, EISDIR...) or a models.json that is a link, and retries a
|
|
749
|
+
// transient errno (issue #552), so the credential gate sees this only when the read failed here and not there.
|
|
750
|
+
// Absent, or not valid JSON, is determinate.
|
|
751
|
+
let modelsUnreadable = null;
|
|
752
|
+
try {
|
|
753
|
+
endpoints = modelEndpoints() ?? [];
|
|
754
|
+
if (endpoints.length > 0) {
|
|
755
|
+
try {
|
|
756
|
+
models = overlayModels();
|
|
757
|
+
} catch (err) {
|
|
758
|
+
// FAIL OPEN, and say so: an overlay models.json that does not parse names no endpoint for any model, and
|
|
759
|
+
// refusing here would refuse every job on a hosted model too. The job runs without an endpoint slot.
|
|
760
|
+
// A FIXED reason, never the error's message: a JSON.parse message quotes the file's text around the fault,
|
|
761
|
+
// and models.json holds keys (PR #518's gate measured one in this line). The settings reader's posture.
|
|
762
|
+
// An fs error is named by its code: `readOverlayModels` returns null for absence and rethrows every other
|
|
763
|
+
// errno as-is (PR #520 round 2), carried for the gate, which retries rather than calling the provider keyless.
|
|
764
|
+
// A configError is determinate: the file was read and is not JSON, or not an object (`[]`), and is named as
|
|
765
|
+
// such, never as "unreadable", which is the fs-error wording.
|
|
766
|
+
const reason = typeof err?.code === "string" ? err.code : err?.overlayLink === true ? "overlay models.json is a link" : err?.overlayNotAFile === true ? "overlay models.json is not a regular file" : /not valid JSON/.test(String(err?.message)) ? "overlay models.json is not valid JSON" : err?.piDispatchConfig === true ? "overlay models.json is not a valid models.json" : "overlay models.json is unreadable";
|
|
767
|
+
deps?.log?.("endpoint_models_unreadable", { jobId: job.id, reason });
|
|
768
|
+
if (typeof err?.code === "string") modelsUnreadable = { code: err.code };
|
|
769
|
+
}
|
|
770
|
+
// ONE hold per endpoint id: a set naming an endpoint twice would take its slot and then wait on itself,
|
|
771
|
+
// forever on `slots: 1`. Deduplicated before the sort, so the order rule sees each endpoint once.
|
|
772
|
+
const byId = new Map();
|
|
773
|
+
for (const e of endpointSetFor({ models, job: effectiveJobOf(job.data, settings, deps?.allowedModels ?? null), endpoints })) if (!byId.has(e.id)) byId.set(e.id, e);
|
|
774
|
+
set = [...byId.values()].sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
|
|
775
|
+
}
|
|
776
|
+
} catch (err) {
|
|
777
|
+
// The derivation reads data only, so this is a defect, not a state: the job runs unbounded and the log says so.
|
|
778
|
+
deps?.log?.("endpoint_gate_unavailable", { jobId: job.id, reason: scrubCredentials(err?.message) });
|
|
779
|
+
set = [];
|
|
780
|
+
}
|
|
781
|
+
endpointSnapshot = { endpoints, models, set, ...(modelsUnreadable ? { modelsUnreadable } : {}) };
|
|
782
|
+
for (const endpoint of set) {
|
|
783
|
+
let where = null;
|
|
784
|
+
let fleet = null;
|
|
785
|
+
if (!endpointSlots.tryAcquire(endpoint.id, endpoint.slots)) {
|
|
786
|
+
where = "host";
|
|
787
|
+
} else if (endpointLease && !isPerMachineHost(endpoint.host)) {
|
|
788
|
+
// A host alias NAME is a DIFFERENT server on every machine (PR #518's gate: two Macs each declaring
|
|
789
|
+
// host.docker.internal shared one fleet bound), so every other host, any address included, takes the
|
|
790
|
+
// fleet half; the in-process bound above is exact for a server only this host reaches.
|
|
791
|
+
fleet = await endpointLease.acquire(job.id, { slots: endpoint.slots, keyArgs: [hash16(endpoint.id)] });
|
|
792
|
+
if (!fleet) {
|
|
793
|
+
endpointSlots.release(endpoint.id);
|
|
794
|
+
where = "fleet";
|
|
795
|
+
} else if (fleet.degraded) {
|
|
796
|
+
deps?.log?.("endpoint_lease_degraded", { jobId: job.id, endpoint: endpoint.id });
|
|
797
|
+
}
|
|
798
|
+
}
|
|
799
|
+
if (where !== null) {
|
|
800
|
+
await releaseEndpointHolds();
|
|
801
|
+
await releaseScopeHolds();
|
|
528
802
|
if (hostHeld) {
|
|
529
803
|
hostBound.slots.release(HOST_SLOT_KEY);
|
|
530
804
|
hostHeld = false;
|
|
531
805
|
}
|
|
532
|
-
deps?.log?.("
|
|
533
|
-
await job.moveToDelayed(nowMs +
|
|
806
|
+
deps?.log?.("endpoint_busy_deferred", { jobId: job.id, endpoint: endpoint.id, where, delayMs: ENDPOINT_BUSY_RECHECK_MS });
|
|
807
|
+
await job.moveToDelayed(nowMs + ENDPOINT_BUSY_RECHECK_MS, token);
|
|
534
808
|
throw new DelayedError();
|
|
535
809
|
}
|
|
810
|
+
endpointHolds.push({ id: endpoint.id, fleet });
|
|
536
811
|
}
|
|
537
812
|
}
|
|
538
813
|
|
|
@@ -691,25 +966,22 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
691
966
|
cancelPoll.unref?.();
|
|
692
967
|
}
|
|
693
968
|
} catch (error) {
|
|
694
|
-
// Release and
|
|
695
|
-
//
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
held = false;
|
|
699
|
-
}
|
|
969
|
+
// Release and DRAIN: this throw never reaches the main finally below, but a shared scope must never be
|
|
970
|
+
// releasable twice -- a double release frees another holder's slot. Last taken, first given back.
|
|
971
|
+
void releaseEndpointHolds();
|
|
972
|
+
void releaseScopeHolds();
|
|
700
973
|
if (hostHeld) {
|
|
701
974
|
hostBound.slots.release(HOST_SLOT_KEY);
|
|
702
975
|
hostHeld = false;
|
|
703
976
|
}
|
|
704
|
-
void scopeSlot?.release?.();
|
|
705
|
-
scopeSlot = null;
|
|
706
977
|
clearTimeout(timer);
|
|
707
978
|
clearInterval(cancelPoll);
|
|
708
979
|
throw error;
|
|
709
980
|
}
|
|
710
981
|
|
|
711
982
|
try {
|
|
712
|
-
|
|
983
|
+
// The read itself happened once, above the endpoint gate; its throw lands HERE, where it always did.
|
|
984
|
+
if (settingsThrew) throw settingsError;
|
|
713
985
|
if (settings.invalid) {
|
|
714
986
|
// A present-but-invalid overlay is a POLICY refusal, RETURNED (never thrown) so BullMQ marks the
|
|
715
987
|
// job completed and does not retry a file that can never parse (CONST-RETRY-INFRA-ONLY). Resolved
|
|
@@ -719,7 +991,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
719
991
|
// refusals are the others): each is decided before or during the settings read, so no honest effective
|
|
720
992
|
// value exists yet -- buildRecord defaults both null.
|
|
721
993
|
const result = { outcome: "policy", reason: "settings-overlay-invalid", exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
722
|
-
|
|
994
|
+
recordAfterGate({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
723
995
|
return result;
|
|
724
996
|
}
|
|
725
997
|
|
|
@@ -729,17 +1001,38 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
729
1001
|
|
|
730
1002
|
// Fill the effective job settings under `job.data > overlay > env` precedence: an explicit per-job
|
|
731
1003
|
// field wins; an omitted one takes the overlay value, else env, resolved at this job's start
|
|
732
|
-
// (INT-CONFIG-OVERLAY-CONTRACT).
|
|
733
|
-
//
|
|
734
|
-
// only after its budget slot is reserved. The `caps`/`softHoldPct` passed to runJob change which
|
|
1004
|
+
// (INT-CONFIG-OVERLAY-CONTRACT). A forge job carries provider/model/maxTurns only when its trigger
|
|
1005
|
+
// named them (#502), so for most jobs this fill supplies the provider the container env allowlist
|
|
1006
|
+
// requires -- absent it, the allowlist refuses a job only after its budget slot is reserved. The `caps`/`softHoldPct` passed to runJob change which
|
|
735
1007
|
// values reserveBudget checks, never when it runs.
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
1008
|
+
// The one call that logs a malformed queued cap (#501): the endpoint gate above computes the same job and
|
|
1009
|
+
// stays quiet, so it is reported once per pickup.
|
|
1010
|
+
const effectiveJob = effectiveJobOf(job.data, settings, deps?.allowedModels ?? null, (event, fields) => deps?.log?.(event, { jobId: job.id, ...fields }));
|
|
1011
|
+
|
|
1012
|
+
// THE ENVELOPE (issue #504 part B, DES-DELEGATED-ALLOCATION-INSIDE-ENVELOPE). With one loaded, every job is governed:
|
|
1013
|
+
// its project's, or `_other`'s, share of the applied split narrows the dollar ledgers below, read ONCE here
|
|
1014
|
+
// beside the limits snapshot and the pickup project, so the gate and the reservation judge one split. The read
|
|
1015
|
+
// also brings the state in line (the neutral seed, expiry, a re-base) and says when this host's envelope is not
|
|
1016
|
+
// the applied split's, which the processor refuses as `envelope-mismatch` before anything is spent. A Valkey
|
|
1017
|
+
// fault throws here, inside the try, and is retried like any reserve fault: nothing has started.
|
|
1018
|
+
const operatorDollars = { dollarCaps: dollarWindowCaps(settings), scopedDollars: dollarCapsFor(job.data, limits), projectDollars: projectDollarCapsFor(limits, project) };
|
|
1019
|
+
let dollarInputs = { ...operatorDollars };
|
|
1020
|
+
const governing = allocation?.current?.() ?? null;
|
|
1021
|
+
try {
|
|
1022
|
+
if (governing?.envelope) {
|
|
1023
|
+
const { state, mismatch } = await allocation.reconcile({ envelope: governing.envelope, digest: governing.digest, now: new Date(nowMs) });
|
|
1024
|
+
dollarInputs = mismatch ? { ...operatorDollars, envelopeMismatch: true } : governedDollars({ envelope: governing.envelope, state, member: project === null ? null : { id: project, member: memberScopeOf(job.data) }, operator: operatorDollars });
|
|
1025
|
+
} else if (typeof allocation?.fleetGoverned === "function" && (await allocation.fleetGoverned())) {
|
|
1026
|
+
// A host with no envelope in a fleet with an applied split: ungoverned, so refused like a differing envelope.
|
|
1027
|
+
dollarInputs = { ...operatorDollars, envelopeMismatch: "no-envelope" };
|
|
1028
|
+
}
|
|
1029
|
+
} catch (error) {
|
|
1030
|
+
// INFRASTRUCTURE, never a verdict (CONST-RETRY-INFRA-ONLY): Valkey did not answer, replied with an error, or the
|
|
1031
|
+
// audit file could not be written. Nothing has been reserved or started, so the job is retried rather than
|
|
1032
|
+
// dropped as the UnrecoverableError a plain throw would become below.
|
|
1033
|
+
deps?.log?.("allocation_read_failed", { jobId: job.id, code: typeof error?.code === "string" ? error.code : "error" });
|
|
1034
|
+
throw new InfraRetry("the allocation state could not be read", { cause: error, reason: "container-never-started", provider: effectiveJob.provider ?? null, model: effectiveJob.model ?? null, budgetReserved: false });
|
|
1035
|
+
}
|
|
743
1036
|
|
|
744
1037
|
const result = await runJob(effectiveJob, {
|
|
745
1038
|
redis,
|
|
@@ -750,11 +1043,45 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
750
1043
|
// The daily TOKEN cap (issue #25), same overlay > env resolution. Check-AFTER, so it gates the
|
|
751
1044
|
// NEXT job on prior recorded spend; null => the daily token counter is disabled.
|
|
752
1045
|
tokenCap: settings.dailyTokenCap,
|
|
753
|
-
// This job's scoped
|
|
754
|
-
//
|
|
755
|
-
//
|
|
756
|
-
|
|
1046
|
+
// This job's scoped job-count ledgers in reserve order (issues #242 and #499 part B): its repo or folder
|
|
1047
|
+
// row's, then its project row's, from the SAME limits snapshot and the SAME pickup project the gate above
|
|
1048
|
+
// read -- one read per pickup, so gate, ledger and record agree for this attempt. Empty when no row carries
|
|
1049
|
+
// a job-count window for this job.
|
|
1050
|
+
scopedLedgers: scopedLedgers(job.data, limits, project),
|
|
1051
|
+
// Issue #501: the deployment's dollar windows in micro-dollars, resolved this job-start under overlay > env
|
|
1052
|
+
// like the caps above, or null when none is set (then nothing is reserved and no dollar key is written).
|
|
1053
|
+
dollarCaps: dollarInputs.dollarCaps,
|
|
1054
|
+
// Issues #501 part 5 and #502 part 6: this job's repo or folder dollar windows and the model dollar windows it
|
|
1055
|
+
// reserves in, from the SAME limits snapshot. The model rows follow the job's EFFECTIVE list (the trigger's,
|
|
1056
|
+
// else PI_ALLOWED_MODELS); a job with none reserves in every model row (`modelDollarRows` says why).
|
|
1057
|
+
scopedDollars: dollarInputs.scopedDollars,
|
|
1058
|
+
// Issue #499 part B: the project row's dollar windows, keyed by the project row's scope, for the pickup project.
|
|
1059
|
+
projectDollars: dollarInputs.projectDollars,
|
|
1060
|
+
// Issue #504 part B: the project was decided on the folder AS NAMED; prepare mounts the
|
|
1061
|
+
// folder it RESOLVES. The processor asks which project the resolved folder belongs to, from the same projects
|
|
1062
|
+
// snapshot, and refuses before any reserve when it is another project's.
|
|
1063
|
+
pickupProject: project,
|
|
1064
|
+
folderProject: (folder) => projectOf({ kind: "local", folder }, pickupProjects),
|
|
1065
|
+
// Issue #504 part B: `_other`'s ledger, the deployment cap's source and the envelope verdict, under an envelope;
|
|
1066
|
+
// absent without one, so the processor's defaults keep such a deployment byte-identical.
|
|
1067
|
+
...(dollarInputs.otherDollars ? { otherDollars: dollarInputs.otherDollars } : {}),
|
|
1068
|
+
...(dollarInputs.dollarCapSource ? { dollarCapSource: dollarInputs.dollarCapSource } : {}),
|
|
1069
|
+
...(dollarInputs.envelopeMismatch ? { envelopeMismatch: dollarInputs.envelopeMismatch } : {}),
|
|
1070
|
+
// Issue #505: this host's envelope delegation block, for the `portfolio-no-envelope` gate; absent without an
|
|
1071
|
+
// envelope, where the processor's default (null, no envelope) is the truth.
|
|
1072
|
+
...(governing?.envelope ? { envelopeDelegation: governing.envelope.delegation ?? null } : {}),
|
|
1073
|
+
modelDollars: modelDollarRows(limits, effectiveJob.models ?? null),
|
|
1074
|
+
// The endpoint gate's snapshot (issue #503): the declared endpoints, the overlay models and this job's
|
|
1075
|
+
// derived set, read once at pickup. Absent on a wiring with no endpoint seam, so a bare processor's
|
|
1076
|
+
// runJob context is unchanged.
|
|
1077
|
+
...(endpointSnapshot ? { modelEndpoints: endpointSnapshot } : {}),
|
|
757
1078
|
...deps,
|
|
1079
|
+
// Issue #502: the skew check must see `models` as the job ARRIVED, not as `effectiveJobOf` filled it from
|
|
1080
|
+
// PI_ALLOWED_MODELS, or a trigger list a stale receiver dropped would read as present and the job would run on
|
|
1081
|
+
// the deployment's list. Only when the wiring supplies the check, so a bare processor keeps the default.
|
|
1082
|
+
// The same for `maxCostUsd` (#501): `effectiveJobOf` keeps the arrived value under its own key and writes the
|
|
1083
|
+
// resolved cap as `maxCostMicros`, but it is read off `job.data` here too, so no later fill can hide a drop.
|
|
1084
|
+
...(deps?.checkWaitSkew ? { checkWaitSkew: (j, ...rest) => deps.checkWaitSkew({ ...j, models: job.data?.models, maxCostUsd: job.data?.maxCostUsd }, ...rest) } : {}),
|
|
758
1085
|
// #227. BOUNDED AFTER THE ABORT, and this is what makes `abortable` an honest declaration.
|
|
759
1086
|
//
|
|
760
1087
|
// `makeRunContainer`'s promise settles ONLY on the docker child's `close` or `error`. Nothing
|
|
@@ -811,6 +1138,9 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
811
1138
|
// mirroring the name/signal injection above. Omitted when unwired so a bare processor falls back
|
|
812
1139
|
// to runJob's no-op default (a chain fault can never flip a completed outcome either way).
|
|
813
1140
|
...(deps.collectChain ? { collectChain: (ctx) => deps.collectChain({ ...ctx, job }) } : {}),
|
|
1141
|
+
// The plan collector (issue #505) the same way and for the same reason: the writer it records is the REAL job's
|
|
1142
|
+
// `.id`, and the authority it checks is `.data` as queued. The pickup's `portfolio` decision rides in `ctx`.
|
|
1143
|
+
...(deps.collectPlan ? { collectPlan: (ctx) => deps.collectPlan({ ...ctx, job }) } : {}),
|
|
814
1144
|
// prepareWorkspace needs the REAL BullMQ job's `.id` to derive a cron job's scheduled-for
|
|
815
1145
|
// instant from the deterministic repeat:<id>:<millis> jobId (DES-CRON-VIA-BULLMQ-SCHEDULER)
|
|
816
1146
|
// for the local /job/event.json. runJob's own `job` is the effectiveJob -- a spread of
|
|
@@ -846,7 +1176,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
846
1176
|
deps.log?.("cancel_acked_before_result", { jobId: job.id, was: `${result?.outcome}/${result?.reason ?? ""}` });
|
|
847
1177
|
outcome = { ...result, outcome: "policy", reason: "operator-cancel" };
|
|
848
1178
|
}
|
|
849
|
-
|
|
1179
|
+
recordAfterGate({ job, result: outcome, startedAt, endedAt: new Date().toISOString() });
|
|
850
1180
|
return outcome;
|
|
851
1181
|
} catch (error) {
|
|
852
1182
|
// Issue #448 (gate round 2 of PR #473): a local job held until rootful Podman's service restarts goes back to the
|
|
@@ -896,10 +1226,10 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
896
1226
|
const endCancelled = async ({ retryable }) => {
|
|
897
1227
|
const spent = error?.budgetReserved === true;
|
|
898
1228
|
const beforeStart = retryable && !spent;
|
|
899
|
-
const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null) };
|
|
1229
|
+
const result = { outcome: "policy", reason: "operator-cancel", exitCode: spent ? (error.exitCode ?? null) : null, turns: spent ? (error.turns ?? null) : null, tokens: spent ? (error.tokens ?? null) : null, ...(spent && error.usage ? { usage: error.usage } : {}), provider: error?.provider ?? null, model: error?.model ?? null, session: error?.session ?? null, budgetReserved: retryable ? spent : (error?.budgetReserved ?? null), ...(error?.dollars ? { dollars: error.dollars } : {}) };
|
|
900
1230
|
deps?.log?.("job_cancelled_instead_of_retry", { jobId: job.id, spent, retryable, ...(retryable ? {} : { failure: scrubCredentials(String(error?.message ?? error)).slice(0, 300) }) });
|
|
901
1231
|
if (deps?.comment) await Promise.resolve(deps.comment(job.data, beforeStart ? CANCELLED_BEFORE_START_COMMENT : TERMINAL_COMMENTS["operator-cancel"])).catch(() => {});
|
|
902
|
-
|
|
1232
|
+
recordAfterGate({ job, result, startedAt, endedAt: new Date().toISOString() });
|
|
903
1233
|
return result;
|
|
904
1234
|
};
|
|
905
1235
|
// STOP THE POLL BEFORE ASKING (gate round 2 of PR #479). The poll ran until the finally below, so a request that
|
|
@@ -932,7 +1262,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
932
1262
|
throw new DelayedError();
|
|
933
1263
|
}
|
|
934
1264
|
const expired = Object.assign(new UnrecoverableError(`held ${Math.round((heldAt - since) / 60_000)} min for rootful Podman's service to restart, and it did not: ${error.message}`), { reason: PODMAN_RESTART_HOLD_EXPIRED, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
935
|
-
|
|
1265
|
+
recordAfterGate({ job, error: expired, startedAt, endedAt: new Date().toISOString() });
|
|
936
1266
|
throw expired;
|
|
937
1267
|
}
|
|
938
1268
|
// Issue #476: a job whose egress preflight found the rootless network keeper running on its own bridge but younger
|
|
@@ -957,14 +1287,14 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
957
1287
|
const was = keeperHold.startedMs ?? young.startedMs;
|
|
958
1288
|
await markKeeperLoop(job, was === young.startedMs ? [was] : [was, young.startedMs]);
|
|
959
1289
|
const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was, now: young.startedMs, heldMs, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
960
|
-
|
|
1290
|
+
recordAfterGate({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
961
1291
|
throw loop;
|
|
962
1292
|
}
|
|
963
1293
|
if (error?.reason === NETNS_KEEPER_NOT_HOLDING && keeperHold.startedMs !== null) {
|
|
964
1294
|
// The keeper this job was waiting on is no longer running on its bridge: it died young, the loop's other face.
|
|
965
1295
|
await markKeeperLoop(job, [keeperHold.startedMs]);
|
|
966
1296
|
const loop = new InfraRetry(netnsKeeperCrashLoopSentence({ was: keeperHold.startedMs, now: null, problem: error.keeperProblem ?? null, heldMs: keeperHold.at - keeperHold.since, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
967
|
-
|
|
1297
|
+
recordAfterGate({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
968
1298
|
throw loop;
|
|
969
1299
|
}
|
|
970
1300
|
// A LATER ATTEMPT OF A JOB THAT SAW THE LOOP (gate of PR #479). The loop's retry comes after the queue's 60 s
|
|
@@ -975,10 +1305,10 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
975
1305
|
const loopSeen = job.data?.netnsKeeperLoopSeen;
|
|
976
1306
|
if (error?.reason === NETNS_KEEPER_NOT_HOLDING && Array.isArray(loopSeen) && loopSeen.length > 0) {
|
|
977
1307
|
const loop = new InfraRetry(netnsKeeperLoopAgainSentence({ seen: loopSeen, problem: error.keeperProblem ?? null, remedy: error.keeperRemedy }), { reason: NETNS_KEEPER_CRASH_LOOP, provider: error.provider ?? null, model: error.model ?? null, budgetReserved: false });
|
|
978
|
-
|
|
1308
|
+
recordAfterGate({ job, error: loop, startedAt, endedAt: new Date().toISOString() });
|
|
979
1309
|
throw loop;
|
|
980
1310
|
}
|
|
981
|
-
|
|
1311
|
+
recordAfterGate({ job, error, startedAt, endedAt: new Date().toISOString() });
|
|
982
1312
|
if (error instanceof InfraRetry) throw error; // retryable: BullMQ retries per attempts
|
|
983
1313
|
// A non-retryable, non-infra error (our bug) must not retry forever. UnrecoverableError
|
|
984
1314
|
// records it as failed-and-distinct in the queue's failed set without a retry.
|
|
@@ -986,14 +1316,18 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
986
1316
|
} finally {
|
|
987
1317
|
// Release FIRST and never throw (release clamps at zero by construction): a throw here would
|
|
988
1318
|
// mask the job's real error, and a missed release wedges the scope until a worker restart.
|
|
989
|
-
|
|
990
|
-
if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
|
|
1319
|
+
// The in-process halves go back synchronously inside each drain, before its first await.
|
|
991
1320
|
// AWAITED, not fire-and-forget. Two reasons, and the second is the one that bites: an unawaited
|
|
992
1321
|
// DEL is dropped by `shutdown`'s `process.exit(0)`, stranding the claim for its whole TTL on a
|
|
993
1322
|
// restart -- and the next same-scope job would otherwise race the release, be denied, and sit out a
|
|
994
1323
|
// full re-check interval while the slot it wanted went free behind it. The finally is already inside
|
|
995
|
-
// an async function, and `release` never throws.
|
|
996
|
-
|
|
1324
|
+
// an async function, and `release` never throws. Last taken, first given back: the endpoint holds
|
|
1325
|
+
// (issue #503), then the scope holds (project, then repo), then the host slot.
|
|
1326
|
+
const endpointsReleased = releaseEndpointHolds();
|
|
1327
|
+
const scopesReleased = releaseScopeHolds();
|
|
1328
|
+
if (hostHeld) hostBound.slots.release(HOST_SLOT_KEY);
|
|
1329
|
+
await endpointsReleased;
|
|
1330
|
+
await scopesReleased;
|
|
997
1331
|
clearTimeout(timer);
|
|
998
1332
|
clearInterval(cancelPoll);
|
|
999
1333
|
signal.removeEventListener("abort", onAbort);
|
|
@@ -1035,7 +1369,7 @@ export function keeperHoldState(job, at) {
|
|
|
1035
1369
|
return { at, since, startedMs };
|
|
1036
1370
|
}
|
|
1037
1371
|
|
|
1038
|
-
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
|
|
1372
|
+
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [] }) {
|
|
1039
1373
|
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
1040
1374
|
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
1041
1375
|
// because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
|
|
@@ -1091,9 +1425,17 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1091
1425
|
// folder mutex, the per-scope ceiling and the wait-check lease all bound the HOST, so two
|
|
1092
1426
|
// independent maps would double every one of them exactly as two Workers double concurrency.
|
|
1093
1427
|
scopedLimits,
|
|
1428
|
+
projects,
|
|
1429
|
+
allocation,
|
|
1094
1430
|
inFlight,
|
|
1095
1431
|
hostBound,
|
|
1096
1432
|
scopeLease,
|
|
1433
|
+
// Issue #503: the model endpoint bound. `endpointSlots` is SHARED across both Workers for the host
|
|
1434
|
+
// slot's reason: it bounds this host's calls on one server, and two maps would double it.
|
|
1435
|
+
endpointSlots,
|
|
1436
|
+
endpointLease,
|
|
1437
|
+
modelEndpoints,
|
|
1438
|
+
overlayModels,
|
|
1097
1439
|
// Issue #230. Undefined pass-throughs take makeProcessor's own defaults (a wait state over the same
|
|
1098
1440
|
// redis client, and the shared 30-day `after` ceiling), so a bare wiring behaves like a wired one.
|
|
1099
1441
|
waitState,
|
|
@@ -1112,6 +1454,9 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1112
1454
|
maxFaults,
|
|
1113
1455
|
deps,
|
|
1114
1456
|
recordRun,
|
|
1457
|
+
// The run-record lookup a stalled job is checked against before it can run again (see the processor's
|
|
1458
|
+
// first gate). `null` in a bare wiring, which keeps today's behaviour: the job runs.
|
|
1459
|
+
settledRecord,
|
|
1115
1460
|
});
|
|
1116
1461
|
|
|
1117
1462
|
// Issue #464: only a connection `parseConnection` built, which judges and pins the Valkey it dials.
|