@edgehero/pi-dispatch 2.0.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +44 -7
- package/README.md +14 -6
- package/deploy/com.pi-dispatch.worker.plist +1 -1
- package/deploy/docker-compose.yml +12 -0
- package/deploy/egress-proxy.conf +28 -3
- package/deploy/pi-dispatch-egress-proxy.container +8 -2
- package/deploy/worker-env-wrapper.cmd +1 -1
- package/deploy/worker-env-wrapper.sh +3 -3
- package/package.json +9 -2
- package/src/allocation.mjs +731 -0
- package/src/backends.mjs +243 -0
- package/src/budget.mjs +40 -4
- package/src/cli.mjs +222 -11
- package/src/config.mjs +126 -5
- package/src/daemon-facts.mjs +3 -0
- package/src/deployment-venue.mjs +1 -0
- package/src/doctor.mjs +2316 -183
- package/src/dollar-budget.mjs +373 -0
- package/src/dollar-fingerprint.mjs +83 -0
- package/src/egress-cli.mjs +316 -0
- package/src/egress-proxy-state.mjs +35 -5
- package/src/egress.mjs +16 -3
- package/src/env-allowlist.mjs +142 -18
- package/src/env-file.mjs +194 -25
- package/src/envelope.mjs +413 -0
- package/src/exit-code.mjs +22 -0
- package/src/fleet-lease.mjs +85 -25
- package/src/get-token.mjs +16 -5
- package/src/git-dirty.mjs +67 -0
- package/src/github-app-setup.mjs +6 -3
- package/src/github-host.mjs +5 -3
- package/src/host-pi.mjs +19 -3
- package/src/identity.mjs +2 -1
- package/src/image-preflight.mjs +98 -24
- package/src/image-ref.mjs +37 -0
- package/src/import-pi.mjs +4 -2
- package/src/index.mjs +407 -62
- package/src/init.mjs +18 -0
- package/src/job-id.mjs +26 -3
- package/src/live-probes.mjs +24 -9
- package/src/model-catalog.mjs +297 -0
- package/src/model-endpoints.mjs +649 -0
- package/src/model-ref.mjs +151 -0
- package/src/models-json.mjs +262 -0
- package/src/money.mjs +144 -0
- package/src/octokit-log.mjs +65 -0
- package/src/outbox-plan.mjs +218 -0
- package/src/outbox.mjs +29 -9
- package/src/output-cap.mjs +157 -0
- package/src/packages.mjs +2 -2
- package/src/pause-windows.mjs +81 -2
- package/src/pi-model-loader.mjs +77 -0
- package/src/podman-stack.mjs +16 -3
- package/src/portfolio-snapshot.mjs +304 -0
- package/src/prepare-local.mjs +247 -12
- package/src/prepare.mjs +35 -3
- package/src/pricing.mjs +9 -5
- package/src/priorities.mjs +569 -0
- package/src/processor.mjs +603 -173
- package/src/project-id.mjs +17 -0
- package/src/projects.mjs +238 -0
- package/src/provider-key.mjs +32 -7
- package/src/provider-steering.mjs +214 -59
- package/src/queue.mjs +111 -6
- package/src/reserved-env.mjs +30 -0
- package/src/run-container.mjs +59 -5
- package/src/run-history.mjs +379 -24
- package/src/run-mirror.mjs +30 -0
- package/src/runtime-settings.mjs +104 -9
- package/src/schedules.mjs +33 -1
- package/src/scoped-limits.mjs +447 -27
- package/src/secrets.mjs +2 -1
- package/src/service.mjs +15 -4
- package/src/session-store.mjs +131 -6
- package/src/start.mjs +528 -40
- package/src/subscriptions.mjs +7 -3
- package/src/triggers-file.mjs +65 -4
- package/src/triggers.mjs +140 -9
- package/src/up.mjs +308 -34
- package/src/valkey-endpoint.mjs +3 -2
package/src/processor.mjs
CHANGED
|
@@ -1,13 +1,25 @@
|
|
|
1
1
|
import { DAEMON_APPLIES_BOUNDS, DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, DOCKER_NEVER_STARTED_EXITS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_CONF_WIDENS_JOB, PODMAN_SERVICE_LOCAL, RUNTIME_ADDS_NO_MOUNTS } from "./backends.mjs";
|
|
2
2
|
import { resolveBackendName } from "./backend-registry.mjs";
|
|
3
3
|
import { lstatSync } from "node:fs";
|
|
4
|
-
import { checkTokenCap, recordTokenSpend,
|
|
4
|
+
import { checkTokenCap, recordTokenSpend, releaseLedgers, reserveLedgers } from "./budget.mjs";
|
|
5
5
|
import { configError } from "./config.mjs";
|
|
6
|
-
import {
|
|
6
|
+
import { unqualifiedScope } from "./pause-windows.mjs";
|
|
7
|
+
import { PROJECT_CAP_REASON } from "./scoped-limits.mjs";
|
|
7
8
|
import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
|
|
9
|
+
import { RESERVED_ENV_NAMES } from "./triggers.mjs";
|
|
8
10
|
import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
|
|
9
|
-
import { RUNNER_POLICY_REASONS } from "./run-history.mjs";
|
|
11
|
+
import { COST_CAP_WHYS, RUNNER_POLICY_REASONS } from "./run-history.mjs";
|
|
10
12
|
import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
|
|
13
|
+
import { CAPABILITY_GATES, EXIT_AUTH_CAPABILITY } from "./image-preflight.mjs";
|
|
14
|
+
import { modelListProblem, modelOnList, splitModelEntry } from "./model-ref.mjs";
|
|
15
|
+
import { ALLOCATION_CAP_REASON, ENVELOPE_MISMATCH_REASON } from "./allocation.mjs";
|
|
16
|
+
|
|
17
|
+
/** The refusal of a local job whose resolved folder belongs to another project than the one decided at pickup (issue #504 part B). */
|
|
18
|
+
export const LOCAL_FOLDER_PROJECT_CHANGED = "local-folder-project-changed";
|
|
19
|
+
// Issue #505: a portfolio cron job refused before it spends, because this host could not apply the plan it would write.
|
|
20
|
+
export const PORTFOLIO_NO_ENVELOPE = "portfolio-no-envelope";
|
|
21
|
+
import { DOLLAR_CAP_REASON, DOLLAR_KEY_PREFIX, dollarLedgers, dollarSettlement, dollarsRecord, holdPart, meteredMicros, modelDollarSettlement, releaseDollars, reserveDollars, settleDollars } from "./dollar-budget.mjs";
|
|
22
|
+
import { zeroRatedVerdict } from "./model-endpoints.mjs";
|
|
11
23
|
|
|
12
24
|
/**
|
|
13
25
|
* The forge comment's reason for each observation a floor refusal missed (issues #278 and #345), keyed like
|
|
@@ -88,8 +100,22 @@ export const TERMINAL_COMMENTS = {
|
|
|
88
100
|
// Issue #437. Names the cause but never the provider's own message, which may echo a key fragment. "Or
|
|
89
101
|
// access" because a 403 is as often a key that works but may not use this model or route as a bad key.
|
|
90
102
|
"provider-auth-refused": "Stopped: the AI provider refused this worker's credentials or access (an authentication or permission error). The operator needs to check the provider key and what it is allowed to use. Not retried.",
|
|
103
|
+
// Issues #501, #502. The two stops name the policy, never the amount or the model: both are operator
|
|
104
|
+
// configuration, and the comment's reader may be an issue author who can act on neither.
|
|
105
|
+
"cost-cap": "Stopped: the next AI call could have taken this run past its cost limit, so it was not made. Partial work may exist. Not retried.",
|
|
106
|
+
"model-not-allowed": "Stopped: the run tried to call an AI model this trigger does not allow, or to change an AI request in a way it does not allow, so the call was not made. Partial work may exist. Not retried.",
|
|
107
|
+
"cost-cap-unenforceable": "Stopped: this run has a cost limit, and the job image could not enforce it before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
|
|
108
|
+
"model-policy-unenforceable": "Stopped: this run is limited to certain AI models, and the job image could not enforce that before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
|
|
91
109
|
};
|
|
92
110
|
|
|
111
|
+
// Issue #502: the `model-unknown` refusal's comment. Names no model: the reader may be an issue author.
|
|
112
|
+
export const MODEL_UNKNOWN_COMMENT = "Refused: this job names an AI model that this deployment does not know (it is in neither pi's model catalog nor the overlay models.json), so no container was started and nothing was spent. Ask the operator to check the trigger's model settings. Not run.";
|
|
113
|
+
|
|
114
|
+
// The `model-unknown` refusal whose `why` is `overlay-*` (PR #558, the end-of-round check of #501 and #502): the deployment's overlay
|
|
115
|
+
// models.json is the problem, not the job's model, so the generic text above would send the author to the wrong
|
|
116
|
+
// place. Fixed text naming no path and no model; the operator finds which file and why in the worker log.
|
|
117
|
+
export const OVERLAY_REFUSED_COMMENT = "Refused before starting: the deployment's model settings file (models.json in the overlay) cannot be used for this job, so no container was started and nothing was spent. The operator needs to fix that file. Not run.";
|
|
118
|
+
|
|
93
119
|
// Issue #341: the forge comments for a `job-user-unmappable` refusal, keyed by cause. Shorter than the operator
|
|
94
120
|
// texts in job-user.mjs on purpose: a comment's reader may be an issue author, who can act on none of it.
|
|
95
121
|
const JOB_USER_COMMENTS = Object.freeze({
|
|
@@ -126,18 +152,54 @@ function egressProxyFix(venue, proxy = DEFAULT_EGRESS_PROXY) {
|
|
|
126
152
|
return "Start it with `pi-dispatch up` from the deployment folder";
|
|
127
153
|
}
|
|
128
154
|
|
|
155
|
+
/**
|
|
156
|
+
* Did the runtime never hand control to the runner (issue #227)? ONE answer for the two places that ask it (issue
|
|
157
|
+
* #501): the exit-code switch, which refunds such an exit as `container-never-started`, and the dollar settlement,
|
|
158
|
+
* which must not settle a container that never ran (the refund in the catch gives its reservation back instead).
|
|
159
|
+
* Two answers could drift: a never-started exit settled at the floor AND refunded would give back twice, and one
|
|
160
|
+
* neither settled nor refunded would leave its hold standing.
|
|
161
|
+
*
|
|
162
|
+
* A detached container (issue #345) DID start, and an aborted one is classified by its flag first, so neither is
|
|
163
|
+
* never-started whatever its code. The runner's own three codes are never the venue's never-started set's to claim.
|
|
164
|
+
*/
|
|
165
|
+
export function isNeverStartedExit({ code, aborted, detached }, neverStartedCodes) {
|
|
166
|
+
if (detached === true || aborted) return false;
|
|
167
|
+
if (code === EXIT_COMPLETED || code === EXIT_POLICY || code === EXIT_INFRA) return false;
|
|
168
|
+
return (neverStartedCodes ?? []).includes(code);
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/** The key prefixes of a job's model dollar ledgers (`modelDollarRows`), to tell a model window's keys in a hold apart. */
|
|
172
|
+
function modelPrefixesOf(modelDollars) {
|
|
173
|
+
return new Set((modelDollars ?? []).map((m) => m.keyPrefix));
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* The models a job may call (issue #501, #503 part 7), as `{ provider, id }` refs: its main model, and every entry of
|
|
178
|
+
* its effective allowed-model list when it has one. A list entry that does not split is kept as `null`, which the
|
|
179
|
+
* zero-rated check reads as "not zero-rated" (fail closed).
|
|
180
|
+
*/
|
|
181
|
+
function callableModelRefs(job) {
|
|
182
|
+
const refs = [{ provider: job.provider, id: job.model }];
|
|
183
|
+
for (const entry of Array.isArray(job.models) ? job.models : []) {
|
|
184
|
+
const ref = splitModelEntry(entry);
|
|
185
|
+
refs.push(ref === null ? null : { provider: ref.provider, id: ref.model });
|
|
186
|
+
}
|
|
187
|
+
return refs;
|
|
188
|
+
}
|
|
189
|
+
|
|
129
190
|
export async function runJob(job, deps) {
|
|
130
191
|
const {
|
|
131
192
|
redis,
|
|
132
193
|
caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
|
|
133
194
|
softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
|
|
134
195
|
tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
|
|
135
|
-
// { scope, caps: { day, week, month } }
|
|
136
|
-
// INT-SCOPED-LIMITS-FILE-CONTRACT)
|
|
137
|
-
//
|
|
138
|
-
//
|
|
139
|
-
//
|
|
140
|
-
|
|
196
|
+
// [{ scope, keyPrefix, caps: { day, week, month }, reason }] -- this job's scoped job-count ledgers in reserve order
|
|
197
|
+
// (issues #242 and #499 part B, INT-SCOPED-LIMITS-FILE-CONTRACT): its repo or folder row's, then its project row's,
|
|
198
|
+
// from ONE builder (`scopedLedgers`) over the same watched-limits snapshot and pickup project the gate read. The
|
|
199
|
+
// global ledger is appended here, last. Empty when no row carries a job-count window; the default keeps an
|
|
200
|
+
// unwired processor byte-identical. The folder MUTEX and the `concurrent` slots do not live here -- they are the
|
|
201
|
+
// pickup gate's, pre-everything; this is only the money half.
|
|
202
|
+
scopedLedgers = [],
|
|
141
203
|
recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
|
|
142
204
|
// (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
|
|
143
205
|
// this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
|
|
@@ -151,13 +213,21 @@ export async function runJob(job, deps) {
|
|
|
151
213
|
// Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
|
|
152
214
|
// unwired seam must not refuse, and the wiring is what turns the check on.
|
|
153
215
|
checkWaitSkew = async () => ({ ok: true }),
|
|
154
|
-
// () => { ok } | { message }. Issue #310. Resolves this deployment's provider credential the way
|
|
216
|
+
// () => { ok } | { message } | { unavailable } (issue #503: a transient overlay read, retried). Issue #310. Resolves this deployment's provider credential the way
|
|
155
217
|
// buildContainerEnv will, and answers whether it exists AT ALL, so an unconfigured provider refuses
|
|
156
218
|
// here rather than inside runContainer with the budget already reserved. Admit-everything by default,
|
|
157
219
|
// like the two above and for their reason. A PROBE, deliberately: it discards whatever it resolves and
|
|
158
220
|
// the real read happens where it always did, because threading a live credential through the processor
|
|
159
221
|
// would put it in scope for every log line and record between here and the container.
|
|
160
222
|
checkProviderCredential = () => ({ ok: true }),
|
|
223
|
+
// (refs) => { ok } | { unknown: { provider, id }, why } | { unavailable: code } (issue #502). Is each of this
|
|
224
|
+
// job's models, its main one and every listed one, a model pi knows (model-catalog.mjs `checkModelsKnown`,
|
|
225
|
+
// the builtin catalog plus the overlay models.json)? Admit-everything by default, like the credential probe
|
|
226
|
+
// above and for its reason: an unwired seam must not refuse, and the wiring is what turns the check on.
|
|
227
|
+
checkModelsKnown = () => ({ ok: true }),
|
|
228
|
+
// Issue #503: the pickup's endpoint snapshot `{ endpoints, models, set }` (index.mjs), read once per pickup. Null on
|
|
229
|
+
// a wiring with no endpoint seam, which leaves the credential gate and the container env exactly as before.
|
|
230
|
+
modelEndpoints = null,
|
|
161
231
|
// REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
|
|
162
232
|
// deployment with no egress policy does -- which is also what the real factory returns when unarmed.
|
|
163
233
|
egressPreflight = async () => ({ ok: true }),
|
|
@@ -230,7 +300,7 @@ export async function runJob(job, deps) {
|
|
|
230
300
|
mintToken,
|
|
231
301
|
isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
|
|
232
302
|
prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
|
|
233
|
-
// runContainer({ job, token, prepared, secrets, name, signal, user, home }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
|
|
303
|
+
// runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
|
|
234
304
|
// `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
|
|
235
305
|
// `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
|
|
236
306
|
// `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
|
|
@@ -247,6 +317,49 @@ export async function runJob(job, deps) {
|
|
|
247
317
|
// or a github job with no /outbox -- chains nothing. It NEVER throws (outbox.mjs), so its counts are
|
|
248
318
|
// additive telemetry that can never flip the parent's completed outcome (CONST-RETRY-INFRA-ONLY).
|
|
249
319
|
collectChain = async () => ({ enqueued: 0, refused: 0 }),
|
|
320
|
+
// The plan collector (issue #505, outbox-plan.mjs): a completed portfolio job's `/outbox/priorities.json`, handed to
|
|
321
|
+
// applyPlan. Called on the completed branch only, after collectChain, with the pickup's `portfolio` decision. It
|
|
322
|
+
// NEVER throws, so a refused plan is a recorded outcome of a completed job and never a retry. The default collects
|
|
323
|
+
// nothing (null: no plan), so a wiring that omits it records `plan: null`.
|
|
324
|
+
collectPlan = async () => null,
|
|
325
|
+
// Issue #501: the deployment's dollar windows `{ day, week, month }` in micro-dollars (each null when unset), or
|
|
326
|
+
// null when no window is set. Null is the default and the off switch: nothing is reserved or settled and no
|
|
327
|
+
// `budget:usd:*` key is written, so a deployment with no dollar setting is byte-identical.
|
|
328
|
+
dollarCaps = null,
|
|
329
|
+
// Issues #501 part 5 and #502 part 6 (scoped-limits.json version 2): this job's repo or folder dollar windows,
|
|
330
|
+
// `{ scope, keyPrefix, caps }` from `dollarCapsFor`, or null; and the model dollar windows it reserves in,
|
|
331
|
+
// `[{ ref, keyPrefix, caps }]` from `modelDollarRows` over its effective list (every model row when it has
|
|
332
|
+
// none). Both default to nothing, so an unwired processor reserves exactly what it did before.
|
|
333
|
+
scopedDollars = null,
|
|
334
|
+
// Issue #499 part B: this job's project row's dollar windows, `{ scope, keyPrefix, caps }` from
|
|
335
|
+
// `projectDollarCapsFor` with the project resolved at pickup, or null. Reserved after the repo or folder row's.
|
|
336
|
+
projectDollars = null,
|
|
337
|
+
modelDollars = [],
|
|
338
|
+
// Issue #504 part B (DES-DELEGATED-ALLOCATION-INSIDE-ENVELOPE): under an envelope, `governedDollars` narrows the
|
|
339
|
+
// ledgers above by the applied split, and these three ride beside them. `otherDollars` is `_other`'s ledger
|
|
340
|
+
// (`{ scope, keyPrefix, caps, capSource }`), for a job in no envelope project; `dollarCapSource` says, per window,
|
|
341
|
+
// whether the deployment cap came from the operator or the envelope total (`scopedDollars` and `projectDollars`
|
|
342
|
+
// carry their own `capSource`). `envelopeMismatch` is the pickup's verdict that this host's envelope is not the
|
|
343
|
+
// applied split's (true), or that it has none while one is applied ("no-envelope"). All default to nothing, so a deployment with no envelope is byte-identical.
|
|
344
|
+
otherDollars = null,
|
|
345
|
+
dollarCapSource = null,
|
|
346
|
+
envelopeMismatch = false,
|
|
347
|
+
// Issue #505: whether the LIVE triggers file still flags this job's cron trigger `run.portfolio: true`, as
|
|
348
|
+
// `(job) => boolean`, and this host's envelope `delegation` block (`{ enabled, writers }`), null with no envelope.
|
|
349
|
+
// The flag on the job data is what the trigger said when the job was queued; the file is what the operator says
|
|
350
|
+
// now, so removing the flag takes effect for a job already queued. The default answers "not flagged": an unwired
|
|
351
|
+
// processor treats every job as an ordinary one, which is what a job with no confirmed flag is.
|
|
352
|
+
checkPortfolioFlag = async () => false,
|
|
353
|
+
envelopeDelegation = null,
|
|
354
|
+
// Issue #504 part B: the project decided at pickup on the folder as named, and a function giving the project of a
|
|
355
|
+
// folder from the same projects snapshot. A local job whose RESOLVED folder belongs to another project is refused
|
|
356
|
+
// after prepare, before any reserve. Absent on a bare wiring, which checks nothing.
|
|
357
|
+
pickupProject = null,
|
|
358
|
+
folderProject = null,
|
|
359
|
+
// Issue #503 part 7: the builtin catalog's model object for (provider, id), or null (model-catalog.mjs
|
|
360
|
+
// `builtinModel`). Read only by the zero-rated check; the default knows no builtin model, so an unwired
|
|
361
|
+
// processor judges overlay models alone and reserves for every other.
|
|
362
|
+
builtinModel = () => null,
|
|
250
363
|
now = new Date(),
|
|
251
364
|
} = deps;
|
|
252
365
|
|
|
@@ -261,10 +374,29 @@ export async function runJob(job, deps) {
|
|
|
261
374
|
const wantsForgeToken = isForgeBacked || job.github === true;
|
|
262
375
|
let token = null;
|
|
263
376
|
let prepared = null;
|
|
264
|
-
|
|
265
|
-
|
|
377
|
+
// The job-count reservations still standing, in reserve order (issue #499 part B): what `reserveLedgers` took and
|
|
378
|
+
// no refund has given back yet. EVERY refund below is `releaseLedgers` over this one list, last first, so no path
|
|
379
|
+
// can give back one ledger and forget another, and none can give one back twice: a released ledger leaves the list.
|
|
380
|
+
const held = [];
|
|
381
|
+
// The global ledger's own entry, so `budgetReserved` can ask the list. ONE rule on every path (a count refusal, a
|
|
382
|
+
// dollar-cap, config-refused, an InfraRetry): `budgetReserved` says whether the GLOBAL slot is still held after any
|
|
383
|
+
// refund, global-only as INT-RUN-HISTORY-FILE-CONTRACT has it. A scoped or project slot a failed refund left behind
|
|
384
|
+
// is in the `budget_release_failed` log line, not in this field.
|
|
385
|
+
let globalLedger = null;
|
|
386
|
+
const globalHeld = () => globalLedger !== null && held.includes(globalLedger);
|
|
387
|
+
// Give back every held job-count slot, NEVER throwing: a refund that did not land must not replace the caller's
|
|
388
|
+
// classification with a Redis message. Logged with `at`; returns whether the list is empty afterwards.
|
|
389
|
+
const refundLedgers = async (at) => {
|
|
390
|
+
try {
|
|
391
|
+
await releaseLedgers(redis, held, { now });
|
|
392
|
+
return true;
|
|
393
|
+
} catch (releaseError) {
|
|
394
|
+
log("budget_release_failed", { at, code: releaseError?.code ?? null });
|
|
395
|
+
return false;
|
|
396
|
+
}
|
|
397
|
+
};
|
|
266
398
|
// Set once `runContainer` has RESOLVED, which is the only moment a container is known to have run. The
|
|
267
|
-
// config classifier in the catch refunds
|
|
399
|
+
// config classifier in the catch refunds every job-count ledger, and its whole justification is that nothing was
|
|
268
400
|
// spent; without a fact to test, that is a claim about where config throws happen to live today rather
|
|
269
401
|
// than a property of the code. A config-tagged throw raised after a paid run would otherwise refund a slot
|
|
270
402
|
// the container really spent AND tell the operator publicly that nothing was.
|
|
@@ -275,6 +407,11 @@ export async function runJob(job, deps) {
|
|
|
275
407
|
// A spawn fault that fails between is an InfraRetry carrying `container-never-started`, refunded by the
|
|
276
408
|
// arm below, which has a discriminator of its own.
|
|
277
409
|
let containerRan = false;
|
|
410
|
+
// Issue #501. `dollarHold` is the dollar reservation still standing (reserveDollars' hold), null when none was
|
|
411
|
+
// taken or once it has been settled or given back, so no path can settle or refund it twice. `dollars` is what
|
|
412
|
+
// the record says about it (INT-RUN-HISTORY-FILE-CONTRACT), null until the reservation step ran.
|
|
413
|
+
let dollarHold = null;
|
|
414
|
+
let dollars = null;
|
|
278
415
|
|
|
279
416
|
try {
|
|
280
417
|
// The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
|
|
@@ -306,6 +443,18 @@ export async function runJob(job, deps) {
|
|
|
306
443
|
// standing between a stale receiver and a paid job that ran when the operator wrote "wait".
|
|
307
444
|
{
|
|
308
445
|
const skew = await checkWaitSkew(job);
|
|
446
|
+
// Issue #502: an authored NARROWING field the job arrived without (`AUTHORED_NARROWING_FIELDS`, triggers-file.mjs),
|
|
447
|
+
// today `run.models` and `run.maxCostUsd`. The same two causes as a dropped wait, and the same refusal shape; without it the job would
|
|
448
|
+
// run on the deployment's list, or on none, while every record reads like a correct run. The FIELD is named,
|
|
449
|
+
// never its value: the comment's reader may be an issue author.
|
|
450
|
+
if (skew.skewed && typeof skew.field === "string") {
|
|
451
|
+
// Three causes, the likeliest first (PR #536's review, round 3, made the check strict): the trigger gained the
|
|
452
|
+
// field after this job was queued, a service is below the version that carries it, or one still reads an
|
|
453
|
+
// older copy of the triggers file. The first is fixed by re-running the job, so the comment says so.
|
|
454
|
+
await comment(job, `Refused: this trigger sets \`run.${skew.field}\`, but the job reached the worker without it, so it would have run without that limit. Either the trigger changed after this job was queued (re-run it), or a service in this deployment is stale: below the version that carries the field, or still reading an older copy of the triggers file and in need of a restart. Not run.`);
|
|
455
|
+
log("refused_trigger_skew", { triggerIndex: job.trigger?.matched?.index ?? null, field: skew.field, causes: "trigger-changed-after-queue-or-stale-service" });
|
|
456
|
+
return { outcome: "policy", reason: "trigger-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
|
|
457
|
+
}
|
|
309
458
|
if (skew.skewed) {
|
|
310
459
|
// Named for the operator, not the payload: how many conditions were authored, never what
|
|
311
460
|
// they say. The fix is a version, so the comment says which one.
|
|
@@ -487,57 +636,16 @@ export async function runJob(job, deps) {
|
|
|
487
636
|
log("refused_image_forge_unsupported", { image: img.forgeUnsupported, kind: img.kind, declared: img.declared });
|
|
488
637
|
return { outcome: "policy", reason: "job-image-forge-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
489
638
|
}
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
await comment(
|
|
501
|
-
job,
|
|
502
|
-
`Refused: the job image "${img.replicaUnsupported}" does not declare replica support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its baked guardrails would name the wrong branch. Rebuild the image from a version that has this feature. Not run.`,
|
|
503
|
-
);
|
|
504
|
-
log("refused_image_replicas_unsupported", { image: img.replicaUnsupported, declared: img.declared });
|
|
505
|
-
return { outcome: "policy", reason: "job-image-replicas-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
506
|
-
}
|
|
507
|
-
if (img.commandUnsupported) {
|
|
508
|
-
// The image is present and does not declare command support (issue #189), so its runner
|
|
509
|
-
// predates run.command: it reads no PI_COMMAND, and the bare `/name args` prompt reaches the
|
|
510
|
-
// model as PROSE -- no handler runs, the agent improvises, and the queue records a clean exit
|
|
511
|
-
// 0. The in-container half of the gate (the runner's own command-unregistered refusal) does
|
|
512
|
-
// not exist on such an image, which is exactly why the host must refuse first.
|
|
513
|
-
//
|
|
514
|
-
// Determinate, so a refusal rather than a retry, and pre-spend, because no version of this
|
|
515
|
-
// gets better by running. Like the replica branch above, the message names the FIX rather
|
|
516
|
-
// than the label that noticed it.
|
|
517
|
-
await comment(
|
|
518
|
-
job,
|
|
519
|
-
`Refused: the job image "${img.commandUnsupported}" does not declare command support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its runner would not dispatch \`run.command\`. Rebuild the image from a version that has this feature. Not run.`,
|
|
520
|
-
);
|
|
521
|
-
log("refused_image_commands_unsupported", { image: img.commandUnsupported, declared: img.declared });
|
|
522
|
-
return { outcome: "policy", reason: "job-image-commands-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
523
|
-
}
|
|
524
|
-
if (img.excludeToolsUnsupported) {
|
|
525
|
-
// The image is present and does not declare exclude-tools support (issue #291), so its runner
|
|
526
|
-
// predates run.excludeTools: it reads no PI_EXCLUDE_TOOLS, and the job would run with every
|
|
527
|
-
// tool the trigger says to remove -- a "read-only" trigger with a working editor and shell,
|
|
528
|
-
// recording a clean exit. That is a PERMISSION quietly not enforced, the silent fail-open this
|
|
529
|
-
// repo brands the worst outcome available, which is exactly why the host refuses before spend
|
|
530
|
-
// rather than letting the container fail open.
|
|
531
|
-
//
|
|
532
|
-
// Determinate, so a refusal rather than a retry, and pre-spend, because no version of this
|
|
533
|
-
// gets better by running. Like the command branch above, the message names the FIX rather
|
|
534
|
-
// than the label that noticed it.
|
|
535
|
-
await comment(
|
|
536
|
-
job,
|
|
537
|
-
`Refused: the job image "${img.excludeToolsUnsupported}" does not declare exclude-tools support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its runner would ignore \`run.excludeTools\` and run this trigger with every tool it says to remove. Rebuild the image from a version that has this feature. Not run.`,
|
|
538
|
-
);
|
|
539
|
-
log("refused_image_exclude_tools_unsupported", { image: img.excludeToolsUnsupported, declared: img.declared });
|
|
540
|
-
return { outcome: "policy", reason: "job-image-exclude-tools-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
639
|
+
// The image capability gates (one table, CAPABILITY_GATES in image-preflight.mjs): the image is present and
|
|
640
|
+
// does not declare a feature this job carries, so its runner would silently ignore it. Determinate, so a
|
|
641
|
+
// refusal rather than a retry, and pre-spend, because no version of this gets better by running. Like the
|
|
642
|
+
// forge branch above, each row's comment names the FIX rather than the label that noticed it.
|
|
643
|
+
const gate = CAPABILITY_GATES.find((row) => img[row.result]);
|
|
644
|
+
if (gate) {
|
|
645
|
+
const image = img[gate.result];
|
|
646
|
+
await comment(job, gate.comment(image, img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"));
|
|
647
|
+
log(gate.event, { image, declared: img.declared });
|
|
648
|
+
return { outcome: "policy", reason: gate.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
541
649
|
}
|
|
542
650
|
if (img.unavailable) {
|
|
543
651
|
// docker itself did not answer -- transient infra, NOT a determinate refusal. THROWN so BullMQ
|
|
@@ -568,6 +676,65 @@ export async function runJob(job, deps) {
|
|
|
568
676
|
throw new InfraRetry("the job user could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
569
677
|
}
|
|
570
678
|
|
|
679
|
+
// Issue #502, two free gates on the job's models, BEFORE the credential gate so a typo in a model or provider
|
|
680
|
+
// id is named as itself rather than as a missing key, and so before the egress probe, the secret resolvers,
|
|
681
|
+
// the mint, the clone, the token-cap read and both reserves (CONST-BUDGET-BEFORE-TOKENS). Both read the
|
|
682
|
+
// EFFECTIVE job (index.mjs `effectiveJobOf`): the main model after the `job.data > overlay > env` fill, and
|
|
683
|
+
// the list as `job.data.models ?? PI_ALLOWED_MODELS`. Checking `job.data` alone would wave through a model
|
|
684
|
+
// the overlay or the env supplied, which is the common case: most forge triggers name none.
|
|
685
|
+
//
|
|
686
|
+
// The model ids are operator configuration, so the log names them; the forge comment does not, for the
|
|
687
|
+
// terminal comments' reason: its reader may be an issue author, who can act on neither.
|
|
688
|
+
{
|
|
689
|
+
// A list that is PRESENT but not a valid list (a string, `{}`, `0`, `false`, `""`, a bad entry) refuses here
|
|
690
|
+
// rather than reading as "no list": the loader and the env parser both refuse such a value, so it is a
|
|
691
|
+
// hand-built or foreign job, and treating it as absent would let it run unrestricted (or hide the
|
|
692
|
+
// deployment's own list behind it). Same refusal as an unknown model, with its own `why`.
|
|
693
|
+
if (job.models !== undefined && job.models !== null && modelListProblem(job.models) !== null) {
|
|
694
|
+
await comment(job, MODEL_UNKNOWN_COMMENT);
|
|
695
|
+
log("refused_model_unknown", { provider: job.provider ?? null, model: job.model ?? null, why: "list-malformed" });
|
|
696
|
+
return { outcome: "policy", reason: "model-unknown", why: "list-malformed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
697
|
+
}
|
|
698
|
+
const refs = [{ provider: job.provider, id: job.model, main: true }];
|
|
699
|
+
for (const entry of Array.isArray(job.models) ? job.models : []) {
|
|
700
|
+
const ref = splitModelEntry(entry);
|
|
701
|
+
refs.push({ provider: ref.provider, id: ref.model });
|
|
702
|
+
}
|
|
703
|
+
const known = await checkModelsKnown(refs);
|
|
704
|
+
if (known?.unavailable) {
|
|
705
|
+
// The overlay models.json could not be READ just now (an errno, never file text). Retried, never
|
|
706
|
+
// refused, the credential gate's rule for the same file below.
|
|
707
|
+
log("model_catalog_unavailable", { provider: job.provider ?? null, model: job.model ?? null, reason: known.unavailable });
|
|
708
|
+
throw new InfraRetry("whether this job's models exist could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
709
|
+
}
|
|
710
|
+
if (known?.unknown) {
|
|
711
|
+
// An `overlay-*` why is the deployment's file, not the job's model: its own comment.
|
|
712
|
+
const why = typeof known.why === "string" ? known.why : null;
|
|
713
|
+
await comment(job, why?.startsWith("overlay-") ? OVERLAY_REFUSED_COMMENT : MODEL_UNKNOWN_COMMENT);
|
|
714
|
+
// `why` is a fixed token: `overlay-unparseable` tells the operator the file is the problem, not the id.
|
|
715
|
+
log("refused_model_unknown", { provider: known.unknown.provider ?? null, model: known.unknown.id ?? null, why: known.why ?? null });
|
|
716
|
+
return { outcome: "policy", reason: "model-unknown", why, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
717
|
+
}
|
|
718
|
+
// The main model must be on the job's list. The loader refuses this when one trigger names all three, so
|
|
719
|
+
// what reaches here is a model or provider the overlay or the env supplied: a `PI_ALLOWED_MODELS` that
|
|
720
|
+
// does not list `PI_MODEL`, or a `dispatch_set model` that moved the default off a trigger's list. Both
|
|
721
|
+
// halves are compared, exact (`modelOnList`), the runner guard's rule.
|
|
722
|
+
if (Array.isArray(job.models) && !modelOnList(job.models, job.provider, job.model)) {
|
|
723
|
+
await comment(job, "Refused: the AI model this job would run on is not on the list of models it is allowed to use, so no container was started and nothing was spent. Ask the operator to check the trigger's model settings. Not run.");
|
|
724
|
+
log("refused_model_not_allowed", { provider: job.provider ?? null, model: job.model ?? null });
|
|
725
|
+
return { outcome: "policy", reason: "model-not-allowed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
726
|
+
}
|
|
727
|
+
// A listed model whose declared fallbacks are not all listed (model-catalog.mjs `declaredFallbacks`): on
|
|
728
|
+
// anthropic-messages pi sends them with every call, so the runner's guard would refuse every call to it after
|
|
729
|
+
// the container started; on any other api the worker is stricter than the runner, by decision. Refused here, free. The log names the listed model, the operator's own configuration; the
|
|
730
|
+
// comment names none.
|
|
731
|
+
if (known?.fallbackUnlisted) {
|
|
732
|
+
await comment(job, "Refused: a model this job is allowed to use declares fallback models that are not on the job's list, so no container was started and nothing was spent. Ask the operator to list those models too, or remove that model. Not run.");
|
|
733
|
+
log("refused_model_not_allowed", { provider: known.fallbackUnlisted.provider ?? null, model: known.fallbackUnlisted.id ?? null, why: "fallback-unlisted" });
|
|
734
|
+
return { outcome: "policy", reason: "model-not-allowed", why: "fallback-unlisted", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
735
|
+
}
|
|
736
|
+
}
|
|
737
|
+
|
|
571
738
|
// Is there a credential to run this job with at all? FREE, determinate and I/O-light: a pure function
|
|
572
739
|
// of the job's provider, the worker env, and (only when the env has no key) one small readFileSync of
|
|
573
740
|
// pi's auth.json. So it goes here, with the other free gates, which is further than issue #310 asked
|
|
@@ -590,7 +757,61 @@ export async function runJob(job, deps) {
|
|
|
590
757
|
// The refusal names no path. `credentialFromPiAuth`'s messages carry auth.json's location, `comment`
|
|
591
758
|
// posts publicly on the issue, and the reason an operator needs is the same either way: their
|
|
592
759
|
// deployment has no usable provider credential and `doctor` will say exactly which variable.
|
|
593
|
-
|
|
760
|
+
// `modelEndpoints` is the pickup's snapshot (issue #503), handed to the gate and to runContainer alike, so a
|
|
761
|
+
// keyless provider passes here and gets its PI_DISPATCH_KEYLESS there from ONE read of the declaration. runContainer
|
|
762
|
+
// is handed it only when an endpoint is declared, so with none its context is byte-identical to before.
|
|
763
|
+
// THE ENVELOPE GATE (issue #504 part B, DES-DELEGATED-ALLOCATION-INSIDE-ENVELOPE): this host's envelope digest is not
|
|
764
|
+
// the applied split's, so the split this job would be judged against was computed for another envelope. FREE and
|
|
765
|
+
// determinate (the pickup read decided it), so it sits with the free gates, before the mint, the clone, the
|
|
766
|
+
// token-cap read and every reserve (CONST-BUDGET-BEFORE-TOKENS), and it RETURNS: a retry meets the same envelope
|
|
767
|
+
// until the operator makes the hosts agree (CONST-RETRY-INFRA-ONLY). Refusing is loud and money-safe; judging one
|
|
768
|
+
// split against two envelopes is neither.
|
|
769
|
+
// `envelopeMismatch` is true (this host's envelope is another one) or "no-envelope" (it has none while the fleet has
|
|
770
|
+
// an applied split): one reason, two texts, since the fix differs.
|
|
771
|
+
if (envelopeMismatch === true || envelopeMismatch === "no-envelope") {
|
|
772
|
+
const absent = envelopeMismatch === "no-envelope";
|
|
773
|
+
await comment(
|
|
774
|
+
job,
|
|
775
|
+
absent
|
|
776
|
+
? "Refused: this worker has no budget envelope while the other workers share an applied budget split, so no container was started and nothing was spent. Ask the operator to install the envelope on this worker, or to turn delegated allocation off for every worker (`pi-dispatch doctor` says how). Not run."
|
|
777
|
+
: "Refused: this worker's budget envelope differs from the one the current budget split was made for, so no container was started and nothing was spent. Ask the operator to run `pi-dispatch doctor`. Not run.",
|
|
778
|
+
);
|
|
779
|
+
log("refused_envelope_mismatch", absent ? { envelope: "none" } : {});
|
|
780
|
+
return { outcome: "policy", reason: ENVELOPE_MISMATCH_REASON, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
// THE PORTFOLIO GATE (issue #505, `portfolio-no-envelope`). A portfolio job exists to write a priorities plan, and a
|
|
784
|
+
// plan applies only on a host whose envelope has delegation on with `portfolio-job` among its writers. Without
|
|
785
|
+
// that the plan is refused after the job has paid to write it, so the job is refused here instead: FREE (the
|
|
786
|
+
// envelope was read at boot or reload, the triggers file is one local read), before the mint, the clone, the
|
|
787
|
+
// token-cap read and every reserve (CONST-BUDGET-BEFORE-TOKENS), and RETURNED, since a retry meets the same
|
|
788
|
+
// envelope (CONST-RETRY-INFRA-ONLY). After the envelope gate, which names the host-wide fault first.
|
|
789
|
+
//
|
|
790
|
+
// The live flag comes FIRST. The job data says what the trigger said when it was queued; the live file says what
|
|
791
|
+
// the operator says now. A job whose trigger no longer flags it (or whose file cannot be read) is an ordinary
|
|
792
|
+
// cron job: it runs unflagged and passes this gate, rather than being refused for a flag nobody holds any more.
|
|
793
|
+
// Only a job with a cron `trigger` and no chain fields is asked about at all: a manual run and a chained child can
|
|
794
|
+
// never be a portfolio job, whatever their data says.
|
|
795
|
+
const portfolio = job.portfolio === true && job.trigger !== undefined && job.parentJobId === undefined && job.chainDepth === undefined && (await Promise.resolve().then(() => checkPortfolioFlag(job)).catch(() => false)) === true;
|
|
796
|
+
if (portfolio) {
|
|
797
|
+
const delegation = envelopeDelegation;
|
|
798
|
+
const why = delegation === null || delegation === undefined ? "no-envelope" : delegation.enabled !== true ? "delegation-off" : !Array.isArray(delegation.writers) || !delegation.writers.includes("portfolio-job") ? "writer-not-allowed" : null;
|
|
799
|
+
if (why !== null) {
|
|
800
|
+
// No comment: a portfolio job is always local, and a local job has no issue to comment on. `why` is a
|
|
801
|
+
// fixed token, so the log says which of the three the operator has to change.
|
|
802
|
+
log("refused_portfolio_no_envelope", { why });
|
|
803
|
+
return { outcome: "policy", reason: PORTFOLIO_NO_ENVELOPE, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
|
|
807
|
+
const credential = await checkProviderCredential(job, { modelEndpoints });
|
|
808
|
+
if (credential.unavailable) {
|
|
809
|
+
// Issue #503: the overlay models.json could not be read at this pickup for a transient reason, so the gate has no
|
|
810
|
+
// verdict for a provider that may be keyless. Retried, never refused: a refusal is permanent and public, and the
|
|
811
|
+
// next attempt may read the file. The code is a fixed errno token, never file text.
|
|
812
|
+
log("provider_credential_unavailable", { provider: job.provider ?? null, reason: credential.unavailable });
|
|
813
|
+
throw new InfraRetry("the provider credential could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
814
|
+
}
|
|
594
815
|
if (!credential.ok) {
|
|
595
816
|
// The PROVIDER is named and the message is not. The provider is operator-authored config, already
|
|
596
817
|
// on the record and the mirror, and named freely by the sibling refusals (`backend-unblessed` names
|
|
@@ -732,7 +953,16 @@ export async function runJob(job, deps) {
|
|
|
732
953
|
//
|
|
733
954
|
// Guarded by `secretsArmed` at the CALL SITE, not inside the resolver: an unflagged job must not reach
|
|
734
955
|
// it under ANY wiring, and a guard that lives in the default is a guard an injected resolver skips.
|
|
735
|
-
|
|
956
|
+
//
|
|
957
|
+
// The load-time reserved names are asked again HERE, of the job itself, for the same reason (issue
|
|
958
|
+
// #511). parseTriggers refuses them when the file loads, but a job does not always come from a file
|
|
959
|
+
// this worker loaded: one queued before an upgrade widened the set, a cron job-scheduler template
|
|
960
|
+
// stored in Valkey (schedules.mjs keeps `run.secrets` in it), or a receiver older than the worker all
|
|
961
|
+
// carry `secrets` the current set would refuse, and buildContainerEnv would write every one of them.
|
|
962
|
+
// The same set the loader uses, imported, so the two cannot drift, and checked before the resolver so
|
|
963
|
+
// no wiring of it can skip the check.
|
|
964
|
+
const loadReserved = secretsArmed(job) ? Object.keys(job.secrets ?? {}).find((name) => RESERVED_ENV_NAMES.has(name)) : undefined;
|
|
965
|
+
const resolved = loadReserved !== undefined ? { reserved: loadReserved, atLoad: true } : secretsArmed(job) ? await resolveSecrets(job) : { ok: true, secrets: {} };
|
|
736
966
|
if (resolved.profileUnknown) {
|
|
737
967
|
await comment(job, "Refused: this trigger set `run.secrets`, and the resolver profile it names is not usable on this worker host. No profile of that name is declared, or its resolver is absent or not executable. The job would have started with those variables unset, and an agent that gets a 401 writes a plausible report and exits 0. Run `pi-dispatch doctor` on the worker to see which profiles it has. Not run.");
|
|
738
968
|
// The operator's own profile LABEL, and never a path, a reference, or a byte the resolver printed.
|
|
@@ -766,11 +996,19 @@ export async function runJob(job, deps) {
|
|
|
766
996
|
// - the worker WRITES the name: buildContainerEnv assigns the provider credential and
|
|
767
997
|
// PI_FORWARD_ENV before this feature's values, so the trigger's value replaces the operator's
|
|
768
998
|
// and every job of that trigger spends the trigger author's key;
|
|
769
|
-
// - the worker does NOT write the name but pi READS it first: the OAuth token variable
|
|
770
|
-
//
|
|
771
|
-
//
|
|
999
|
+
// - the worker does NOT write the name but pi READS it first: the OAuth token variable, and from
|
|
1000
|
+
// the 0.99.1 pin the bearer ANTHROPIC_AUTH_TOKEN (issue #509), are deliberately never written
|
|
1001
|
+
// (apiKeyVariable skips both), so a trigger binding one lands beside the operator's key and
|
|
1002
|
+
// outranks it in pi's own precedence.
|
|
772
1003
|
// The old message asserted the first for both, which is exactly backwards for the second.
|
|
773
|
-
|
|
1004
|
+
// A name the triggers file itself would refuse at load (issue #511) reaches here only from a job
|
|
1005
|
+
// queued before that refusal, a stored cron scheduler template, or an older receiver, so it says that.
|
|
1006
|
+
await comment(
|
|
1007
|
+
job,
|
|
1008
|
+
resolved.atLoad
|
|
1009
|
+
? `Refused: this job's \`run.secrets\` binds \`${resolved.reserved}\`, a name the triggers file refuses at load because this deployment writes it or pi reads it to configure a provider or itself. The job was queued before that refusal applied, by a stored cron schedule, or by an older receiver. Rename it in the triggers file. Not run.`
|
|
1010
|
+
: `Refused: this trigger's \`run.secrets\` binds \`${resolved.reserved}\`, which is a variable this deployment already uses for the job's own credentials. Whichever of the two values reached the container, one of them would be silently ignored. Rename it in the triggers file. Not run.`,
|
|
1011
|
+
);
|
|
774
1012
|
// The variable NAME only. It is the operator's own choice of name, not payload, and naming it is what
|
|
775
1013
|
// makes the refusal actionable -- but the REFERENCE behind it never appears.
|
|
776
1014
|
log("refused_secret_name_reserved", { kind: job.kind ?? null, name: resolved.reserved });
|
|
@@ -796,7 +1034,7 @@ export async function runJob(job, deps) {
|
|
|
796
1034
|
// than by matching its stderr, which image-preflight.mjs forbids for good reason. Folding this into
|
|
797
1035
|
// the refusal above would permanently burn a delivery over a twenty-second vault blip, and a webhook
|
|
798
1036
|
// does not redeliver itself. Nothing has spent: `budgetReserved` computes false in the catch below
|
|
799
|
-
// because
|
|
1037
|
+
// because no ledger is held yet.
|
|
800
1038
|
log("secret_resolver_unreachable", { kind: job.kind ?? null, name: resolved.unreachable, failure: resolved.failure ?? null, code: resolved.code ?? null, stderrBytes: resolved.stderrBytes ?? 0 });
|
|
801
1039
|
throw new InfraRetry(`secret resolver could not answer for ${resolved.unreachable}`, { reason: "secret-resolver-unreachable", provider: job.provider ?? null, model: job.model ?? null });
|
|
802
1040
|
}
|
|
@@ -827,16 +1065,35 @@ export async function runJob(job, deps) {
|
|
|
827
1065
|
}
|
|
828
1066
|
|
|
829
1067
|
// `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
|
|
830
|
-
|
|
1068
|
+
// `portfolio` (issue #505) only for a job the gate above confirmed: prepare then writes /job/portfolio.json, after
|
|
1069
|
+
// asking the live file once more. Absent otherwise, so every other job's prepare call is unchanged.
|
|
1070
|
+
prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
|
|
831
1071
|
|
|
832
1072
|
// A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
|
|
833
1073
|
// or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
|
|
834
1074
|
// -- is POLICY: return before reserveBudget so it burns no cap slot and is never retried.
|
|
835
1075
|
// Mirrors the branch-protection policy return above. Spread-plus-attribution: the prepare
|
|
836
1076
|
// result keeps its own reason and fields, and the host-effective provider/model land beside
|
|
837
|
-
// them exactly as on every other terminal result.
|
|
1077
|
+
// them exactly as on every other terminal result. `budgetReserved: false` like every pre-reserve refusal (issue
|
|
1078
|
+
// #507): nothing was reserved and no container started, so the record says so and the cost fold counts the run as
|
|
1079
|
+
// an exact $0 rather than a floor (REQ-COST-ANALYTICS (d)). After the spread, so no preparer can say otherwise.
|
|
838
1080
|
if (prepared?.outcome === "policy") {
|
|
839
|
-
return { ...prepared, provider: job.provider ?? null, model: job.model ?? null };
|
|
1081
|
+
return { ...prepared, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
|
|
1082
|
+
}
|
|
1083
|
+
|
|
1084
|
+
// THE RESOLVED FOLDER'S PROJECT (issue #504 part B). The pickup decided the project, and so the share a job reserves
|
|
1085
|
+
// against, from the folder AS NAMED; the container mounts the folder prepare RESOLVED. A link or a case variant
|
|
1086
|
+
// inside a run root that leads into ANOTHER project's member would bill that work to the named project's share.
|
|
1087
|
+
// Refused here, after prepare and before the token-cap read and every reserve: free, determinate, never retried. A
|
|
1088
|
+
// resolved folder in no project is not refused, because the named spelling is the operator's own membership
|
|
1089
|
+
// (a symlinked member is listed by the path the triggers use, docs/projects.md).
|
|
1090
|
+
if (job.kind === "local" && typeof folderProject === "function" && typeof prepared?.workspace === "string") {
|
|
1091
|
+
const resolvedProject = folderProject(prepared.workspace);
|
|
1092
|
+
if (resolvedProject !== null && resolvedProject !== pickupProject) {
|
|
1093
|
+
await comment(job, "Refused: this job's folder now leads into another project's folder, so no container was started and nothing was spent. Not run.");
|
|
1094
|
+
log("refused_local_folder_project_changed", { pickup: pickupProject, resolved: resolvedProject });
|
|
1095
|
+
return { outcome: "policy", reason: LOCAL_FOLDER_PROJECT_CHANGED, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
1096
|
+
}
|
|
840
1097
|
}
|
|
841
1098
|
|
|
842
1099
|
// Daily TOKEN cap (issue #25): the deliberate check-AFTER control. Token cost is only known
|
|
@@ -856,75 +1113,169 @@ export async function runJob(job, deps) {
|
|
|
856
1113
|
return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
857
1114
|
}
|
|
858
1115
|
|
|
859
|
-
//
|
|
860
|
-
//
|
|
861
|
-
//
|
|
862
|
-
//
|
|
863
|
-
//
|
|
864
|
-
//
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
|
|
874
|
-
|
|
875
|
-
|
|
876
|
-
|
|
877
|
-
|
|
878
|
-
|
|
879
|
-
|
|
880
|
-
|
|
881
|
-
|
|
882
|
-
|
|
883
|
-
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
1116
|
+
// The JOB-COUNT ledgers (issues #242 and #499 part B, INT-SCOPED-LIMITS-FILE-CONTRACT), narrowest first: the repo
|
|
1117
|
+
// or folder row, the project row, then the global windows, reserved in that order by ONE helper. A narrow
|
|
1118
|
+
// ledger's refusal never consumes a slot in a wider one -- the global INCR runs only for jobs every scoped ledger
|
|
1119
|
+
// admitted, and a project's only for jobs their repo admitted. Same atomic INCR, same refused-still-counts
|
|
1120
|
+
// invariant per ledger, through budget.mjs's keyPrefix seam (budget:s:<hash16> of the row scope). softHoldPct is
|
|
1121
|
+
// deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob; scoped windows
|
|
1122
|
+
// are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
|
|
1123
|
+
//
|
|
1124
|
+
// A redis fault mid-walk gives back every LEDGER that landed whole (`held` is exact) and rethrows. Inside the
|
|
1125
|
+
// ledger that faulted, the windows INCRed before the fault stay counted (a week INCR that faults leaves that
|
|
1126
|
+
// ledger's day counted, an EXPIRE that faults leaves its key without a TTL): `reserveBudget` does not say which of
|
|
1127
|
+
// its windows landed, so giving them back could DECR a window that never rose. That is the pre-existing
|
|
1128
|
+
// mid-reserve posture of one ledger, unchanged.
|
|
1129
|
+
globalLedger = { scope: null, keyPrefix: null, caps, softHoldPct, reason: null };
|
|
1130
|
+
let counted;
|
|
1131
|
+
try {
|
|
1132
|
+
counted = await reserveLedgers(redis, [...(scopedLedgers ?? []), globalLedger], held, { now });
|
|
1133
|
+
} catch (error) {
|
|
1134
|
+
// Valkey failed mid-walk. No container can have started, and `held` lists exactly the reservations that
|
|
1135
|
+
// landed, so they go back (last first, never throwing) before the error escapes; otherwise a repo and a
|
|
1136
|
+
// project slot would stay counted for a job that never ran.
|
|
1137
|
+
await refundLedgers("reserve-fault");
|
|
1138
|
+
throw error;
|
|
1139
|
+
}
|
|
1140
|
+
if (!counted.allowed) {
|
|
1141
|
+
// Every ledger BEFORE the refusing one gives its slot back, last first: a ledger that did not issue the
|
|
1142
|
+
// refusal gives back. Without this, an exhausted global window drains every arriving scope's and project's own
|
|
1143
|
+
// counters with zero runs to show for it, and a full project drains its members' repo windows. The refusing
|
|
1144
|
+
// ledger keeps its own slot (refused-still-counts, per ledger).
|
|
1145
|
+
await refundLedgers(counted.refusedBy.reason ?? counted.result.reason);
|
|
1146
|
+
const result = counted.result;
|
|
1147
|
+
const w = result.blockedWindow;
|
|
1148
|
+
const win = result.windows[w];
|
|
1149
|
+
if (counted.refusedBy === globalLedger) {
|
|
1150
|
+
if (result.reason === "soft-hold") {
|
|
1151
|
+
await comment(job, `Soft-hold: ${w} spend ${win.reserved}/${win.cap} is inside the ${softHoldPct}% hold band. New starts paused; not run.`);
|
|
1152
|
+
log("soft_hold", { window: w, reserved: win.reserved, cap: win.cap, pct: softHoldPct });
|
|
1153
|
+
} else {
|
|
1154
|
+
await comment(job, `Over the ${w} budget cap (${win.cap}). Not run.`);
|
|
1155
|
+
log("over_budget", { window: w, reserved: win.reserved, cap: win.cap });
|
|
1156
|
+
}
|
|
1157
|
+
// budgetReserved true: the global slot is reserved and kept (a refused reservation still counts). Both
|
|
1158
|
+
// over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
|
|
1159
|
+
return { outcome: "policy", reason: result.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
|
|
888
1160
|
}
|
|
1161
|
+
const isProject = counted.refusedBy.reason === PROJECT_CAP_REASON;
|
|
1162
|
+
// A local job's scope is a full host path and its "comment" is not dropped -- the wiring's local adapter LOGS
|
|
1163
|
+
// the text (start.mjs forgeFor fallthrough) -- so the path must never enter the message; "this folder" is enough
|
|
1164
|
+
// beside the jobId the adapter logs. A forge scope IS the repo the comment posts on, safe to name, and named
|
|
1165
|
+
// without a forge prefix (issue #498). A project is "this project": the comment's reader may be an issue
|
|
1166
|
+
// author, and which repos an operator groups is the operator's business, not theirs.
|
|
1167
|
+
const scopeLabel = isProject ? "this project" : job.kind === "local" ? "this folder" : unqualifiedScope(counted.refusedBy.scope);
|
|
1168
|
+
await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
|
|
1169
|
+
// The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would put a full host
|
|
1170
|
+
// path in the worker log against no-pii-in-logs. The admin recomputes the key from the configured scope. A
|
|
1171
|
+
// project refusal adds `ledger: "project"`; a repo or folder one keeps the line it always had.
|
|
1172
|
+
log("over_scope_budget", { scopeKey: counted.refusedBy.keyPrefix, ...(isProject ? { ledger: "project" } : {}), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
|
|
1173
|
+
// budgetReserved false: the GLOBAL slot was never touched (every scoped ledger reserves first).
|
|
1174
|
+
return { outcome: "policy", reason: counted.refusedBy.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
889
1175
|
}
|
|
890
1176
|
|
|
891
|
-
//
|
|
892
|
-
//
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
1177
|
+
// DOLLAR windows (issue #501, part 3), after EVERY job-count reserve and before the container: the last gate
|
|
1178
|
+
// before the spend line (CONST-BUDGET-BEFORE-TOKENS). Every free gate and every job-count ledger come first, so
|
|
1179
|
+
// a refusal anywhere above never touches a dollar counter. The amount is the job's per-job cost cap, which the
|
|
1180
|
+
// runner enforces before every call, so it bounds what the run can spend; the worker needs no prices.
|
|
1181
|
+
let containerJob = job;
|
|
1182
|
+
// Every dollar ledger this job reserves in: the deployment's windows, its repo or folder row's, and each model
|
|
1183
|
+
// row it may reach (scoped-limits.json version 2). ONE reservation over all of them, so a refusal in any window
|
|
1184
|
+
// gives back every key, the deployment's included.
|
|
1185
|
+
const ledgers = dollarLedgers(dollarCaps, { scope: scopedDollars, project: projectDollars, other: otherDollars, models: modelDollars });
|
|
1186
|
+
const modelPrefixes = new Map((modelDollars ?? []).map((m) => [m.keyPrefix, m.ref]));
|
|
1187
|
+
// The ledgers that settle to the JOB's cost (the deployment's, the repo or folder row's, the project row's), named
|
|
1188
|
+
// rather than derived as "not a model", so a ledger kind added later settles nowhere until it is put on a list.
|
|
1189
|
+
// `_other`'s ledger (issue #504 part B) settles to the job's cost too: its counter is what the unassigned work spent.
|
|
1190
|
+
const jobCostPrefixes = new Set([DOLLAR_KEY_PREFIX, scopedDollars?.keyPrefix, projectDollars?.keyPrefix, otherDollars?.keyPrefix].filter((p) => typeof p === "string"));
|
|
1191
|
+
if (ledgers.length > 0) {
|
|
1192
|
+
// A window needs a per-job cap (the settings invariant, `checkDollarInvariant`; for a scoped or model row, the
|
|
1193
|
+
// same rule), so a job with none here is a defect, a hand-built queue entry, or a dollar row in
|
|
1194
|
+
// scoped-limits.json on a deployment with no `maxCostUsd`: refused as configuration, before anything is
|
|
1195
|
+
// reserved, and the config arm below refunds every job-count slot.
|
|
1196
|
+
if (job.maxCostMicros === null || job.maxCostMicros === undefined) throw configError("a dollar window is set but this job has no per-job cost cap to reserve");
|
|
1197
|
+
// Issue #503 part 7: a job that CANNOT spend reserves nothing. Every model it may call is served by a declared
|
|
1198
|
+
// endpoint and zero-rated, so its container runs under a per-job cap of 0, which the runner's cost guard holds
|
|
1199
|
+
// before every call: a call whose bound is above 0 is refused, so the job's spend is 0 by construction, not
|
|
1200
|
+
// by trust in the cost table. The capability gate above already required `costCap` of this job's image (the
|
|
1201
|
+
// job carried a non-null cap), so an image that would ignore the 0 never gets here. A cap that is ALREADY 0
|
|
1202
|
+
// (a malformed queued value reads as 0, `effectiveCostCapMicros`) has nothing to reserve either.
|
|
1203
|
+
const refs = callableModelRefs(job);
|
|
1204
|
+
const zero = job.maxCostMicros === 0 ? { zeroRated: true } : zeroRatedVerdict({ models: modelEndpoints?.models ?? null, endpoints: modelEndpoints?.endpoints ?? [], refs, builtinModel });
|
|
1205
|
+
if (zero.zeroRated) {
|
|
1206
|
+
containerJob = { ...job, maxCostMicros: 0 };
|
|
1207
|
+
dollars = dollarsRecord({ reservedMicros: 0, settledMicros: 0, basis: "unreserved" });
|
|
1208
|
+
log("dollar_unreserved", { models: refs.length });
|
|
915
1209
|
} else {
|
|
916
|
-
|
|
917
|
-
|
|
1210
|
+
let reservation;
|
|
1211
|
+
try {
|
|
1212
|
+
reservation = await reserveDollars(redis, { ledgers, amountMicros: job.maxCostMicros, now, log });
|
|
1213
|
+
} catch (error) {
|
|
1214
|
+
// Valkey did not answer. reserveDollars tried to give back what it had added, key by key (a key it could
|
|
1215
|
+
// not is logged, dollar_giveback_error), so nothing is held by this job; the job-count
|
|
1216
|
+
// slots are refunded by the never-started arm below, and the job is retried: nothing started.
|
|
1217
|
+
log("dollar_reserve_error", { code: typeof error?.code === "string" ? error.code : "error" });
|
|
1218
|
+
throw new InfraRetry("the dollar windows could not be reserved", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
1219
|
+
}
|
|
1220
|
+
if (!reservation.allowed) {
|
|
1221
|
+
// Refused, and every dollar key it touched given back, best effort per key (the departure from the job-count
|
|
1222
|
+
// rule, see dollar-budget.mjs; a key that could not be is logged and the record says floor). Every held
|
|
1223
|
+
// job-count slot goes back too, last first: a ledger that did not issue the refusal gives back, the rule
|
|
1224
|
+
// the count ledgers follow among themselves above.
|
|
1225
|
+
// WHOSE NUMBER bound (issue #504 part B): `allocation-cap` when the refusing window's cap came from the applied split
|
|
1226
|
+
// or the envelope total (`capSource`), `dollar-cap` when it was the operator's own (a tie is the operator's). One
|
|
1227
|
+
// ledger and one window refused, and the reservation names both.
|
|
1228
|
+
const capSourceOf = (prefix) => (prefix === DOLLAR_KEY_PREFIX ? dollarCapSource : [scopedDollars, projectDollars, otherDollars].find((l) => l?.keyPrefix === prefix)?.capSource);
|
|
1229
|
+
const byAllocation = capSourceOf(reservation.ledger)?.[reservation.window] === "allocation";
|
|
1230
|
+
const refusalReason = byAllocation ? ALLOCATION_CAP_REASON : DOLLAR_CAP_REASON;
|
|
1231
|
+
const refunded = await refundLedgers(refusalReason);
|
|
1232
|
+
// The window is named and the amounts are not: the comment's reader may be an issue author, and the
|
|
1233
|
+
// amounts are the operator's, which the log carries. Which ledger refused is named too: the deployment,
|
|
1234
|
+
// the repo (a forge scope IS the repo the comment posts on), "this folder" (a local scope is a host path,
|
|
1235
|
+
// kept out of the comment and the log), or the model (operator configuration, never payload). The log
|
|
1236
|
+
// names a scoped or model ledger by its key prefix, a hash, never the scope string.
|
|
1237
|
+
const period = { day: "today's", week: "this week's", month: "this month's" }[reservation.window] ?? "a";
|
|
1238
|
+
// Two different states, told apart (PR #549's review): a window whose cap is BELOW one job's cap refuses every
|
|
1239
|
+
// such job until the operator changes a setting, which "no room left" would hide behind a wait that never ends.
|
|
1240
|
+
const capBelowJob = reservation.capMicros < job.maxCostMicros;
|
|
1241
|
+
// A project window (issue #499 part B) is "this project", for the job-count refusal's reason.
|
|
1242
|
+
const refusedBy = reservation.ledger === DOLLAR_KEY_PREFIX ? "deployment" : modelPrefixes.has(reservation.ledger) ? "model" : reservation.ledger === projectDollars?.keyPrefix ? "project" : reservation.ledger === otherDollars?.keyPrefix ? "other" : "scope";
|
|
1243
|
+
const whose = refusedBy === "deployment" ? "this deployment" : refusedBy === "model" ? `the model ${modelPrefixes.get(reservation.ledger)}` : refusedBy === "project" ? "this project" : refusedBy === "other" ? "the work outside the budget split's projects" : job.kind === "local" ? "this folder" : unqualifiedScope(scopedDollars.scope);
|
|
1244
|
+
// The allocation's own words (issue #504 part B): the number that bound is the split inside the operator's envelope,
|
|
1245
|
+
// which moves with the next priorities plan or an envelope edit, so this text never tells its reader to raise a budget.
|
|
1246
|
+
const adjective = { day: "daily", week: "weekly", month: "monthly" }[reservation.window] ?? "";
|
|
1247
|
+
const allocationText = capBelowJob
|
|
1248
|
+
? `Refused: the ${adjective} share of the budget split for ${whose} is smaller than this run's cost limit, so no container was started and nothing was spent. The split changes with the next priorities plan or an envelope change. Not run.`
|
|
1249
|
+
: `Refused: ${period} share of the budget split for ${whose} has no room left for this run's cost limit, so no container was started and nothing was spent. Not run.`;
|
|
1250
|
+
await comment(
|
|
1251
|
+
job,
|
|
1252
|
+
byAllocation
|
|
1253
|
+
? allocationText
|
|
1254
|
+
: capBelowJob
|
|
1255
|
+
? `Refused: the ${{ day: "daily", week: "weekly", month: "monthly" }[reservation.window] ?? ""} dollar budget for ${whose} is smaller than this run's cost limit, so no run with this limit can start until the operator raises the budget or lowers the limit. No container was started and nothing was spent. Not run.`
|
|
1256
|
+
: `Refused: ${period} dollar budget for ${whose} has no room left for this run's cost limit, so no container was started and nothing was spent. Not run.`,
|
|
1257
|
+
);
|
|
1258
|
+
log("over_dollar_budget", { ledger: refusedBy, ...(refusedBy === "deployment" ? {} : { key: reservation.ledger }), ...(refusedBy === "model" ? { model: modelPrefixes.get(reservation.ledger) } : {}), window: reservation.window, reservedMicros: reservation.reservedMicros, capMicros: reservation.capMicros, amountMicros: job.maxCostMicros, ...(capBelowJob ? { capBelowJob: true } : {}), ...(byAllocation ? { source: "allocation" } : {}), refunded });
|
|
1259
|
+
const stranded = reservation.stranded > 0;
|
|
1260
|
+
const modelBasis = modelPrefixes.size === 0 ? null : stranded ? "floor" : "refunded";
|
|
1261
|
+
return { outcome: "policy", reason: refusalReason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: globalHeld(), dollars: stranded ? dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: job.maxCostMicros, basis: "floor", modelBasis }) : dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: 0, basis: "refunded", modelBasis }) }; // return => not retried
|
|
1262
|
+
}
|
|
1263
|
+
dollarHold = reservation.hold;
|
|
1264
|
+
dollars = dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: job.maxCostMicros, basis: "floor", modelBasis: modelPrefixes.size > 0 ? "floor" : null });
|
|
918
1265
|
}
|
|
919
|
-
// budgetReserved true: the slot is reserved above and kept (a refused reservation still counts). Both
|
|
920
|
-
// over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
|
|
921
|
-
return { outcome: "policy", reason: budget.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
|
|
922
1266
|
}
|
|
923
1267
|
|
|
924
1268
|
// The user the gate above decided is the user that runs: one answer, never two call sites that agree.
|
|
925
|
-
|
|
1269
|
+
// Issue #545: an image that declares `exitAuth` is handed a per-job key on stdin and signs its exit line with it, and
|
|
1270
|
+
// then only a signed line is read (run-container.mjs, run-history.mjs `authenticExitLines`). An image that does not
|
|
1271
|
+
// declare it is read as before, under the #542 trust rule below alone.
|
|
1272
|
+
const exitAuth = (img.capabilities ?? []).includes(EXIT_AUTH_CAPABILITY);
|
|
1273
|
+
const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}) });
|
|
926
1274
|
containerRan = true;
|
|
927
|
-
|
|
1275
|
+
// `exitAuth: "unverified"` is a run whose image signs its exit line and no signed line was found: the runner died
|
|
1276
|
+
// before writing one, or a line was forged or taken off the pipe. Its tokens read as unknown and its dollars settle
|
|
1277
|
+
// at the floor, the same as a container that wrote no exit line at all.
|
|
1278
|
+
log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}) });
|
|
928
1279
|
|
|
929
1280
|
// Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
|
|
930
1281
|
// so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
|
|
@@ -937,12 +1288,53 @@ export async function runJob(job, deps) {
|
|
|
937
1288
|
await recordSpend(redis, tokensSpent, { now }).catch((err) => log("token_spend_error", { reason: err?.message }));
|
|
938
1289
|
}
|
|
939
1290
|
|
|
1291
|
+
// SETTLE the dollar reservation (issue #501, part 4), ONCE, here, for every exit where the container ran:
|
|
1292
|
+
// completed, runner policy, a worker abort or cancel, exit 1, an unknown exit, and a detached container. A
|
|
1293
|
+
// never-started exit is not settled: the catch refunds it whole, and `isNeverStartedExit` is the one answer both
|
|
1294
|
+
// ask. Before any classification below, so every return and every throw carries the same `dollars`.
|
|
1295
|
+
const neverStarted = isNeverStartedExit({ code, aborted, detached }, neverStartedExits(job));
|
|
1296
|
+
if (dollarHold !== null && !neverStarted) {
|
|
1297
|
+
// The exit line is trusted only when the container exited ON ITS OWN (the worker did not abort, cancel, time
|
|
1298
|
+
// out or detach it, so the runner had the chance to write its genuine last line) AND that line's own `code` is
|
|
1299
|
+
// the container's real exit code (PR #542's review, round 3: a job's tool can forge a $0 line before a stop).
|
|
1300
|
+
const trusted = !aborted && detached !== true && Number.isSafeInteger(exitLineCode) && exitLineCode === code;
|
|
1301
|
+
const reservedMicros = dollarHold.amountMicros;
|
|
1302
|
+
const { settledMicros, basis } = dollarSettlement({ tokens, usage: usage ?? null, reservedMicros, trusted });
|
|
1303
|
+
// The deployment, the repo or folder and the project windows settle to the job's cost; each model window to its
|
|
1304
|
+
// own row of the usage ledger (`modelDollarSettlement`), never to the job's total.
|
|
1305
|
+
const jobPart = holdPart(dollarHold, (prefix) => jobCostPrefixes.has(prefix));
|
|
1306
|
+
const { applied, of } = await settleDollars(redis, jobPart, settledMicros, { log });
|
|
1307
|
+
let modelBasis = null;
|
|
1308
|
+
for (const part of dollarHold.ledgers ?? []) {
|
|
1309
|
+
const ref = modelPrefixes.get(part.keyPrefix);
|
|
1310
|
+
if (ref === undefined) continue;
|
|
1311
|
+
const m = modelDollarSettlement({ ref, basis, tokens, usage: usage ?? null, reservedMicros, trusted });
|
|
1312
|
+
const done = await settleDollars(redis, { amountMicros: reservedMicros, keys: part.keys }, m.settledMicros, { log });
|
|
1313
|
+
// The deployment rule below, per model window: a fault that adjusted none of its keys left the reservation.
|
|
1314
|
+
const mBasis = done.applied === 0 && done.of > 0 && m.settledMicros !== reservedMicros ? "floor" : m.basis;
|
|
1315
|
+
modelBasis = modelBasis === "floor" || mBasis === "floor" ? "floor" : "metered";
|
|
1316
|
+
// The model is operator configuration (a scoped-limits row), and the key a hash; no value but the amounts.
|
|
1317
|
+
log("dollar_model_settled", { model: ref, key: part.keyPrefix, settledMicros: m.settledMicros, basis: mBasis });
|
|
1318
|
+
}
|
|
1319
|
+
// A fault that adjusted no key leaves the whole reservation in every window, which is a floor whatever the
|
|
1320
|
+
// basis would have been, and the record says so. A partial one is logged (dollar_settle_error) and recorded
|
|
1321
|
+
// as computed: the keys that were adjusted hold the settled amount.
|
|
1322
|
+
dollars = applied === 0 && of > 0 && settledMicros !== reservedMicros ? dollarsRecord({ reservedMicros, settledMicros: reservedMicros, basis: "floor", modelBasis }) : dollarsRecord({ reservedMicros, settledMicros, basis, modelBasis });
|
|
1323
|
+
dollarHold = null;
|
|
1324
|
+
log("dollar_settled", { reservedMicros: dollars.reservedMicros, settledMicros: dollars.settledMicros, basis: dollars.basis, ...(modelBasis !== null ? { modelBasis } : {}) });
|
|
1325
|
+
} else if (dollars?.basis === "unreserved") {
|
|
1326
|
+
// Nothing was held, so nothing is written. A metered cost above 0 here would mean the runner's cap of 0 let a
|
|
1327
|
+
// priced call through, which it cannot by construction; it is logged so it cannot pass unseen.
|
|
1328
|
+
const spent = meteredMicros(tokens?.cost);
|
|
1329
|
+
if (spent !== null && spent > 0) log("dollar_unreserved_spent", { meteredMicros: spent });
|
|
1330
|
+
}
|
|
1331
|
+
|
|
940
1332
|
// Issue #345: `docker run` exited with a never-started code, but THIS attempt's container was found by its cidfile,
|
|
941
1333
|
// still there, and was stopped and removed (run-container.mjs, measured on Podman with its API service killed
|
|
942
1334
|
// mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
|
|
943
1335
|
// BEFORE the exit-code switch, where the same code would read as a free never-started exit.
|
|
944
1336
|
if (detached === true) {
|
|
945
|
-
throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
1337
|
+
throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
946
1338
|
}
|
|
947
1339
|
|
|
948
1340
|
// A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
|
|
@@ -960,7 +1352,7 @@ export async function runJob(job, deps) {
|
|
|
960
1352
|
// Awaited bare like every determinate refusal above: the adapter never throws by contract, and
|
|
961
1353
|
// the one swallowed comment in this file (the catch's) justifies itself by its position.
|
|
962
1354
|
await comment(job, TERMINAL_COMMENTS[reason]);
|
|
963
|
-
return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
|
|
1355
|
+
return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}) };
|
|
964
1356
|
}
|
|
965
1357
|
|
|
966
1358
|
switch (code) {
|
|
@@ -971,6 +1363,10 @@ export async function runJob(job, deps) {
|
|
|
971
1363
|
// over-budget, infra): an InfraRetry job is retried, so chaining there would double-enqueue.
|
|
972
1364
|
// collectChain never throws; chainEnqueued/chainRefused are additive telemetry only.
|
|
973
1365
|
const chain = await collectChain({ job, prepared });
|
|
1366
|
+
// Issue #505: the plan, after the chain and on this branch only, for the reason the chain is here: a policy or
|
|
1367
|
+
// infra exit collects nothing, and a retried job must not apply what its failed attempt wrote. `portfolio` is
|
|
1368
|
+
// the pickup's decision; the collector asks the live file again, and both must agree. Never throws.
|
|
1369
|
+
const plan = await collectPlan({ job, prepared, portfolio });
|
|
974
1370
|
// COMPLETED-ONLY PROMOTION, and the exclusivity is the point rather than an optimisation.
|
|
975
1371
|
// A policy or infra exit leaves the canonical transcript byte-identical to what it was
|
|
976
1372
|
// before this run, so a retry starts from exactly what the first attempt did -- promote on
|
|
@@ -994,6 +1390,10 @@ export async function runJob(job, deps) {
|
|
|
994
1390
|
budgetReserved: true,
|
|
995
1391
|
chainEnqueued: chain.enqueued,
|
|
996
1392
|
chainRefused: chain.refused,
|
|
1393
|
+
...(dollars ? { dollars } : {}),
|
|
1394
|
+
// Only when the collector returned a plan (a file, or a confirmed portfolio job with none, `plan-absent`): the
|
|
1395
|
+
// record's `plan` is null otherwise, and every other result is unchanged.
|
|
1396
|
+
...(plan ? { plan } : {}),
|
|
997
1397
|
};
|
|
998
1398
|
}
|
|
999
1399
|
case EXIT_POLICY: {
|
|
@@ -1016,7 +1416,11 @@ export async function runJob(job, deps) {
|
|
|
1016
1416
|
// Every other reason the runner gives still reads as runner-policy.
|
|
1017
1417
|
const reason = RUNNER_POLICY_REASONS.has(exitReason) && code === 2 ? exitReason : "runner-policy";
|
|
1018
1418
|
await comment(job, TERMINAL_COMMENTS[reason]);
|
|
1019
|
-
|
|
1419
|
+
// Issue #507: which rule of the cost guard refused (`unboundable`, `external`, `over-cap`), as the record's
|
|
1420
|
+
// `why`. parseExitWhy keeps only a member of the closed COST_CAP_WHYS off a `cost-cap` line that said code 2,
|
|
1421
|
+
// and it rides only beside a `cost-cap` reason, so a forged line can at worst name the wrong rule of three.
|
|
1422
|
+
const why = reason === "cost-cap" && COST_CAP_WHYS.includes(exitWhy) ? { why: exitWhy } : {};
|
|
1423
|
+
return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why };
|
|
1020
1424
|
}
|
|
1021
1425
|
case EXIT_INFRA:
|
|
1022
1426
|
// NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
|
|
@@ -1024,7 +1428,7 @@ export async function runJob(job, deps) {
|
|
|
1024
1428
|
// the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
|
|
1025
1429
|
// BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
|
|
1026
1430
|
// function never sees (issue #288).
|
|
1027
|
-
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
1431
|
+
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
1028
1432
|
default:
|
|
1029
1433
|
// THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
|
|
1030
1434
|
// (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
|
|
@@ -1038,10 +1442,11 @@ export async function runJob(job, deps) {
|
|
|
1038
1442
|
// silently wrong for any venue where 125 is a real runner exit, and the assumption was
|
|
1039
1443
|
// invisible while there was one runtime. An adapter declares its own set, or declares none
|
|
1040
1444
|
// and normalises to this outcome itself.
|
|
1041
|
-
if (
|
|
1445
|
+
if (neverStarted) {
|
|
1446
|
+
// No `dollars` here: the hold is still standing, and the catch refunds it whole with the job-count slots.
|
|
1042
1447
|
throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
1043
1448
|
}
|
|
1044
|
-
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
1449
|
+
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
|
|
1045
1450
|
}
|
|
1046
1451
|
} catch (e) {
|
|
1047
1452
|
// A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
|
|
@@ -1058,35 +1463,50 @@ export async function runJob(job, deps) {
|
|
|
1058
1463
|
// this project exists to make legible.
|
|
1059
1464
|
//
|
|
1060
1465
|
// Refunding is right BECAUSE no container started: identical to `container-never-started`, and the
|
|
1061
|
-
// same
|
|
1062
|
-
// double-release. The gate above catches the provider case for free, before the mint and the clone;
|
|
1466
|
+
// same all-or-none refund of every held job-count ledger, through the same `releaseLedgers` list, so it still
|
|
1467
|
+
// cannot double-release. The gate above catches the provider case for free, before the mint and the clone;
|
|
1063
1468
|
// this is the backstop for every other config throw that can still land here (an unknown forge kind in
|
|
1064
1469
|
// `buildContainerEnv`, a prepare-time refusal), which would otherwise keep the same slot silently.
|
|
1065
1470
|
// `!containerRan` is the discriminator, and it is what makes the refund and the sentence below TRUE
|
|
1066
1471
|
// rather than merely true today. Every config-tagged throw site in the worker is pre-container, so this
|
|
1067
1472
|
// changes nothing now; the day one is added after a paid run, that run keeps its slot and falls through
|
|
1068
1473
|
// to the untagged path instead of being refunded and publicly declared free.
|
|
1474
|
+
// Issue #501: a dollar hold still standing here was neither settled nor refunded. Two kinds of throw give it back
|
|
1475
|
+
// whole, because no container ran: never-started (the arm below) and config-refused (the next arm). Any other
|
|
1476
|
+
// throw may have followed a container that ran (runContainer itself throwing, a defect), so the hold STAYS, the
|
|
1477
|
+
// floor, which errs toward overcounting, and `dollar_hold_unsettled` says so. Released FIRST and never throwing
|
|
1478
|
+
// (releaseDollars), so a Valkey fault cannot replace either arm's classification.
|
|
1479
|
+
if (dollarHold !== null) {
|
|
1480
|
+
const hold = dollarHold;
|
|
1481
|
+
dollarHold = null;
|
|
1482
|
+
if ((e?.piDispatchConfig === true && !containerRan) || isNeverStartedRetry(e)) {
|
|
1483
|
+
const { applied, of } = await releaseDollars(redis, hold, { log });
|
|
1484
|
+
const heldModel = (hold.ledgers ?? []).some((l) => modelPrefixesOf(modelDollars).has(l.keyPrefix));
|
|
1485
|
+
dollars = dollarsRecord({ reservedMicros: hold.amountMicros, settledMicros: applied === of ? 0 : hold.amountMicros, basis: applied === of ? "refunded" : "floor", modelBasis: heldModel ? (applied === of ? "refunded" : "floor") : null });
|
|
1486
|
+
} else {
|
|
1487
|
+
log("dollar_hold_unsettled", { reservedMicros: hold.amountMicros, error: e instanceof InfraRetry ? "infra-retry" : (e?.name ?? "error") });
|
|
1488
|
+
const heldModel = (hold.ledgers ?? []).some((l) => modelPrefixesOf(modelDollars).has(l.keyPrefix));
|
|
1489
|
+
dollars = dollarsRecord({ reservedMicros: hold.amountMicros, settledMicros: hold.amountMicros, basis: "floor", modelBasis: heldModel ? "floor" : null });
|
|
1490
|
+
}
|
|
1491
|
+
}
|
|
1492
|
+
// The record reads `dollars` off the error on every throw path (buildRecord), so a retried or failed attempt says
|
|
1493
|
+
// what its windows were charged.
|
|
1494
|
+
if (dollars !== null && e !== null && typeof e === "object") {
|
|
1495
|
+
try {
|
|
1496
|
+
e.dollars = dollars;
|
|
1497
|
+
} catch {
|
|
1498
|
+
// a frozen error keeps its own fields; the log lines above are the record of the hold
|
|
1499
|
+
}
|
|
1500
|
+
}
|
|
1501
|
+
|
|
1069
1502
|
if (e?.piDispatchConfig === true && !containerRan) {
|
|
1070
|
-
// GUARDED, and the
|
|
1503
|
+
// GUARDED, and the `held` list is the record of what actually happened. `releaseLedgers` is a loop of
|
|
1071
1504
|
// DECRs over the active windows and can reject part-way (a read-only replica, a dropped
|
|
1072
1505
|
// connection), which would otherwise replace this determinate refusal with a Redis message: the
|
|
1073
1506
|
// operator would be told their queue is broken when their deployment is misconfigured, and the
|
|
1074
1507
|
// escaping error is untagged so it is not retried either. A refund that did not land must not be
|
|
1075
1508
|
// reported as one, so `budgetReserved` follows the ledger and not the intent.
|
|
1076
|
-
|
|
1077
|
-
try {
|
|
1078
|
-
if (reserved) {
|
|
1079
|
-
await releaseBudget(redis, { caps, now });
|
|
1080
|
-
reserved = false;
|
|
1081
|
-
}
|
|
1082
|
-
if (scopedReserved && scopedCaps) {
|
|
1083
|
-
await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
1084
|
-
scopedReserved = false;
|
|
1085
|
-
}
|
|
1086
|
-
} catch (releaseError) {
|
|
1087
|
-
refunded = false;
|
|
1088
|
-
log("budget_release_failed", { at: "config-refused", code: releaseError?.code ?? null });
|
|
1089
|
-
}
|
|
1509
|
+
const refunded = await refundLedgers("config-refused");
|
|
1090
1510
|
// A FIXED sentence, and NO message in the log either. Two different reasons, both load-bearing:
|
|
1091
1511
|
// `credentialFromPiAuth` puts `auth.json`'s location in its refusals and `prepare-local` puts the
|
|
1092
1512
|
// operator's folder in its own, which `buildRecord` reduces to a basename precisely because a host
|
|
@@ -1102,26 +1522,24 @@ export async function runJob(job, deps) {
|
|
|
1102
1522
|
// shipped adapter never throws; this makes that a property of the arm rather than of the wiring.
|
|
1103
1523
|
await comment(job, "Refused: this deployment is misconfigured, so the job could not be started. Ask the operator to run `pi-dispatch doctor`. Not run.").catch(() => {});
|
|
1104
1524
|
log("refused_config", { kind: job.kind ?? null, refunded });
|
|
1105
|
-
// budgetReserved reflects the LEDGER:
|
|
1106
|
-
|
|
1107
|
-
return { outcome: "policy", reason: "config-refused", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: !refunded }; // return => not retried
|
|
1525
|
+
// budgetReserved reflects the LEDGER: whether the GLOBAL slot is still held after the refund (`globalHeld`).
|
|
1526
|
+
return { outcome: "policy", reason: "config-refused", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: globalHeld(), ...(dollars ? { dollars } : {}) }; // return => not retried
|
|
1108
1527
|
}
|
|
1109
1528
|
|
|
1110
1529
|
// A spawn fault (docker daemon down / binary missing) reserved a slot but never started a
|
|
1111
1530
|
// container, so nothing was spent -- give the slot back before the retry. Every other throw
|
|
1112
1531
|
// here (exit-1 infra, unknown exit) means the container ran and legitimately spent its slot,
|
|
1113
|
-
// so `reason` gates the release to the never-started case only.
|
|
1114
|
-
// once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
// scoped one precedes the global one, and the container follows both), so they refund
|
|
1121
|
-
// together -- and a scoped refusal returned above without ever touching the global ledger.
|
|
1122
|
-
if (reserved) await releaseBudget(redis, { caps, now });
|
|
1123
|
-
if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
|
|
1532
|
+
// so `reason` gates the release to the never-started case only. It releases only what `held` still
|
|
1533
|
+
// lists and run once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
|
|
1534
|
+
if (isNeverStartedRetry(e)) {
|
|
1535
|
+
// All-or-none (issues #242 and #499 part B): a never-started container follows EVERY job-count reserve, so
|
|
1536
|
+
// every held ledger refunds together, last first, through the one helper -- and a scoped or project refusal
|
|
1537
|
+
// returned above with the ledgers before it already given back.
|
|
1538
|
+
await refundLedgers("container-never-started");
|
|
1124
1539
|
}
|
|
1540
|
+
// After the refund, so it says what the ledger holds: false when never-started gave the global slot back, true
|
|
1541
|
+
// for a real container that ran and spent (exit-1 infra / unknown exit) or a refund that did not land.
|
|
1542
|
+
if (e instanceof InfraRetry) e.budgetReserved = globalHeld();
|
|
1125
1543
|
throw e;
|
|
1126
1544
|
} finally {
|
|
1127
1545
|
if (prepared) await cleanup(prepared).catch(() => {});
|
|
@@ -1197,8 +1615,17 @@ const SECRET_FAILURES = {
|
|
|
1197
1615
|
nul: "printed a value containing a NUL byte, which cannot survive the container's argv",
|
|
1198
1616
|
};
|
|
1199
1617
|
|
|
1618
|
+
/**
|
|
1619
|
+
* Is this throw the "the runner never ran" retry (issue #227), whichever path raised it: a never-started exit
|
|
1620
|
+
* (`isNeverStartedExit`), a spawn fault, or a pre-start gate? The catch's refunds (the job-count slots and, issue
|
|
1621
|
+
* #501, the dollar hold) key off this one test.
|
|
1622
|
+
*/
|
|
1623
|
+
function isNeverStartedRetry(e) {
|
|
1624
|
+
return e instanceof InfraRetry && e.reason === "container-never-started";
|
|
1625
|
+
}
|
|
1626
|
+
|
|
1200
1627
|
export class InfraRetry extends Error {
|
|
1201
|
-
constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved } = {}) {
|
|
1628
|
+
constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars } = {}) {
|
|
1202
1629
|
super(message, cause ? { cause } : undefined);
|
|
1203
1630
|
this.name = "InfraRetry";
|
|
1204
1631
|
this.piDispatchRetry = true;
|
|
@@ -1217,6 +1644,9 @@ export class InfraRetry extends Error {
|
|
|
1217
1644
|
this.provider = provider ?? null;
|
|
1218
1645
|
this.model = model ?? null;
|
|
1219
1646
|
this.budgetReserved = budgetReserved ?? null;
|
|
1647
|
+
// Issue #501: the dollar reservation's outcome (`dollarsRecord`), or null when no dollar window applied. Set by
|
|
1648
|
+
// the processor on a throw after the reservation, so a retried attempt's record says what its window was charged.
|
|
1649
|
+
this.dollars = dollars ?? null;
|
|
1220
1650
|
}
|
|
1221
1651
|
}
|
|
1222
1652
|
|