@edgehero/pi-dispatch 2.1.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/.env.example +41 -5
  2. package/README.md +11 -5
  3. package/deploy/docker-compose.yml +12 -0
  4. package/deploy/egress-proxy.conf +28 -3
  5. package/deploy/pi-dispatch-egress-proxy.container +8 -2
  6. package/package.json +8 -1
  7. package/src/allocation.mjs +731 -0
  8. package/src/backends.mjs +243 -0
  9. package/src/budget.mjs +40 -4
  10. package/src/cli.mjs +222 -11
  11. package/src/config.mjs +126 -5
  12. package/src/daemon-facts.mjs +3 -0
  13. package/src/deployment-venue.mjs +1 -0
  14. package/src/doctor.mjs +2261 -203
  15. package/src/dollar-budget.mjs +373 -0
  16. package/src/dollar-fingerprint.mjs +83 -0
  17. package/src/egress-cli.mjs +316 -0
  18. package/src/egress-proxy-state.mjs +35 -5
  19. package/src/egress.mjs +12 -0
  20. package/src/env-allowlist.mjs +107 -6
  21. package/src/env-file.mjs +194 -25
  22. package/src/envelope.mjs +413 -0
  23. package/src/exit-code.mjs +22 -0
  24. package/src/fleet-lease.mjs +85 -25
  25. package/src/get-token.mjs +16 -5
  26. package/src/git-dirty.mjs +67 -0
  27. package/src/github-app-setup.mjs +6 -3
  28. package/src/github-host.mjs +5 -3
  29. package/src/identity.mjs +2 -1
  30. package/src/image-preflight.mjs +98 -24
  31. package/src/image-ref.mjs +37 -0
  32. package/src/import-pi.mjs +4 -2
  33. package/src/index.mjs +407 -62
  34. package/src/init.mjs +18 -0
  35. package/src/job-id.mjs +26 -3
  36. package/src/live-probes.mjs +24 -9
  37. package/src/model-catalog.mjs +297 -0
  38. package/src/model-endpoints.mjs +649 -0
  39. package/src/model-ref.mjs +151 -0
  40. package/src/models-json.mjs +262 -0
  41. package/src/money.mjs +144 -0
  42. package/src/octokit-log.mjs +65 -0
  43. package/src/outbox-plan.mjs +218 -0
  44. package/src/outbox.mjs +29 -9
  45. package/src/output-cap.mjs +157 -0
  46. package/src/pause-windows.mjs +81 -2
  47. package/src/pi-model-loader.mjs +77 -0
  48. package/src/podman-stack.mjs +16 -3
  49. package/src/portfolio-snapshot.mjs +304 -0
  50. package/src/prepare-local.mjs +247 -12
  51. package/src/prepare.mjs +35 -3
  52. package/src/priorities.mjs +569 -0
  53. package/src/processor.mjs +599 -170
  54. package/src/project-id.mjs +17 -0
  55. package/src/projects.mjs +238 -0
  56. package/src/provider-steering.mjs +179 -65
  57. package/src/queue.mjs +111 -6
  58. package/src/reserved-env.mjs +30 -0
  59. package/src/run-container.mjs +59 -5
  60. package/src/run-history.mjs +379 -24
  61. package/src/run-mirror.mjs +30 -0
  62. package/src/runtime-settings.mjs +104 -9
  63. package/src/schedules.mjs +33 -1
  64. package/src/scoped-limits.mjs +447 -27
  65. package/src/service.mjs +15 -4
  66. package/src/session-store.mjs +131 -6
  67. package/src/start.mjs +528 -40
  68. package/src/triggers-file.mjs +65 -4
  69. package/src/triggers.mjs +135 -7
  70. package/src/up.mjs +308 -34
  71. package/src/valkey-endpoint.mjs +3 -2
package/src/processor.mjs CHANGED
@@ -1,13 +1,25 @@
1
1
  import { DAEMON_APPLIES_BOUNDS, DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, DOCKER_NEVER_STARTED_EXITS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_CONF_WIDENS_JOB, PODMAN_SERVICE_LOCAL, RUNTIME_ADDS_NO_MOUNTS } from "./backends.mjs";
2
2
  import { resolveBackendName } from "./backend-registry.mjs";
3
3
  import { lstatSync } from "node:fs";
4
- import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
4
+ import { checkTokenCap, recordTokenSpend, releaseLedgers, reserveLedgers } from "./budget.mjs";
5
5
  import { configError } from "./config.mjs";
6
- import { scopeKeyPrefix } from "./scoped-limits.mjs";
6
+ import { unqualifiedScope } from "./pause-windows.mjs";
7
+ import { PROJECT_CAP_REASON } from "./scoped-limits.mjs";
7
8
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
9
+ import { RESERVED_ENV_NAMES } from "./triggers.mjs";
8
10
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
9
- import { RUNNER_POLICY_REASONS } from "./run-history.mjs";
11
+ import { COST_CAP_WHYS, RUNNER_POLICY_REASONS } from "./run-history.mjs";
10
12
  import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
13
+ import { CAPABILITY_GATES, EXIT_AUTH_CAPABILITY } from "./image-preflight.mjs";
14
+ import { modelListProblem, modelOnList, splitModelEntry } from "./model-ref.mjs";
15
+ import { ALLOCATION_CAP_REASON, ENVELOPE_MISMATCH_REASON } from "./allocation.mjs";
16
+
17
+ /** The refusal of a local job whose resolved folder belongs to another project than the one decided at pickup (issue #504 part B). */
18
+ export const LOCAL_FOLDER_PROJECT_CHANGED = "local-folder-project-changed";
19
+ // Issue #505: a portfolio cron job refused before it spends, because this host could not apply the plan it would write.
20
+ export const PORTFOLIO_NO_ENVELOPE = "portfolio-no-envelope";
21
+ import { DOLLAR_CAP_REASON, DOLLAR_KEY_PREFIX, dollarLedgers, dollarSettlement, dollarsRecord, holdPart, meteredMicros, modelDollarSettlement, releaseDollars, reserveDollars, settleDollars } from "./dollar-budget.mjs";
22
+ import { zeroRatedVerdict } from "./model-endpoints.mjs";
11
23
 
12
24
  /**
13
25
  * The forge comment's reason for each observation a floor refusal missed (issues #278 and #345), keyed like
@@ -88,8 +100,22 @@ export const TERMINAL_COMMENTS = {
88
100
  // Issue #437. Names the cause but never the provider's own message, which may echo a key fragment. "Or
89
101
  // access" because a 403 is as often a key that works but may not use this model or route as a bad key.
90
102
  "provider-auth-refused": "Stopped: the AI provider refused this worker's credentials or access (an authentication or permission error). The operator needs to check the provider key and what it is allowed to use. Not retried.",
103
+ // Issues #501, #502. The two stops name the policy, never the amount or the model: both are operator
104
+ // configuration, and the comment's reader may be an issue author who can act on neither.
105
+ "cost-cap": "Stopped: the next AI call could have taken this run past its cost limit, so it was not made. Partial work may exist. Not retried.",
106
+ "model-not-allowed": "Stopped: the run tried to call an AI model this trigger does not allow, or to change an AI request in a way it does not allow, so the call was not made. Partial work may exist. Not retried.",
107
+ "cost-cap-unenforceable": "Stopped: this run has a cost limit, and the job image could not enforce it before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
108
+ "model-policy-unenforceable": "Stopped: this run is limited to certain AI models, and the job image could not enforce that before each AI call, so nothing was sent to the AI provider. The operator needs to update the job image. Not retried.",
91
109
  };
92
110
 
111
+ // Issue #502: the `model-unknown` refusal's comment. Names no model: the reader may be an issue author.
112
+ export const MODEL_UNKNOWN_COMMENT = "Refused: this job names an AI model that this deployment does not know (it is in neither pi's model catalog nor the overlay models.json), so no container was started and nothing was spent. Ask the operator to check the trigger's model settings. Not run.";
113
+
114
+ // The `model-unknown` refusal whose `why` is `overlay-*` (PR #558, the end-of-round check of #501 and #502): the deployment's overlay
115
+ // models.json is the problem, not the job's model, so the generic text above would send the author to the wrong
116
+ // place. Fixed text naming no path and no model; the operator finds which file and why in the worker log.
117
+ export const OVERLAY_REFUSED_COMMENT = "Refused before starting: the deployment's model settings file (models.json in the overlay) cannot be used for this job, so no container was started and nothing was spent. The operator needs to fix that file. Not run.";
118
+
93
119
  // Issue #341: the forge comments for a `job-user-unmappable` refusal, keyed by cause. Shorter than the operator
94
120
  // texts in job-user.mjs on purpose: a comment's reader may be an issue author, who can act on none of it.
95
121
  const JOB_USER_COMMENTS = Object.freeze({
@@ -126,18 +152,54 @@ function egressProxyFix(venue, proxy = DEFAULT_EGRESS_PROXY) {
126
152
  return "Start it with `pi-dispatch up` from the deployment folder";
127
153
  }
128
154
 
155
+ /**
156
+ * Did the runtime never hand control to the runner (issue #227)? ONE answer for the two places that ask it (issue
157
+ * #501): the exit-code switch, which refunds such an exit as `container-never-started`, and the dollar settlement,
158
+ * which must not settle a container that never ran (the refund in the catch gives its reservation back instead).
159
+ * Two answers could drift: a never-started exit settled at the floor AND refunded would give back twice, and one
160
+ * neither settled nor refunded would leave its hold standing.
161
+ *
162
+ * A detached container (issue #345) DID start, and an aborted one is classified by its flag first, so neither is
163
+ * never-started whatever its code. The runner's own three codes are never the venue's never-started set's to claim.
164
+ */
165
+ export function isNeverStartedExit({ code, aborted, detached }, neverStartedCodes) {
166
+ if (detached === true || aborted) return false;
167
+ if (code === EXIT_COMPLETED || code === EXIT_POLICY || code === EXIT_INFRA) return false;
168
+ return (neverStartedCodes ?? []).includes(code);
169
+ }
170
+
171
+ /** The key prefixes of a job's model dollar ledgers (`modelDollarRows`), to tell a model window's keys in a hold apart. */
172
+ function modelPrefixesOf(modelDollars) {
173
+ return new Set((modelDollars ?? []).map((m) => m.keyPrefix));
174
+ }
175
+
176
+ /**
177
+ * The models a job may call (issue #501, #503 part 7), as `{ provider, id }` refs: its main model, and every entry of
178
+ * its effective allowed-model list when it has one. A list entry that does not split is kept as `null`, which the
179
+ * zero-rated check reads as "not zero-rated" (fail closed).
180
+ */
181
+ function callableModelRefs(job) {
182
+ const refs = [{ provider: job.provider, id: job.model }];
183
+ for (const entry of Array.isArray(job.models) ? job.models : []) {
184
+ const ref = splitModelEntry(entry);
185
+ refs.push(ref === null ? null : { provider: ref.provider, id: ref.model });
186
+ }
187
+ return refs;
188
+ }
189
+
129
190
  export async function runJob(job, deps) {
130
191
  const {
131
192
  redis,
132
193
  caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
133
194
  softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
134
195
  tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
135
- // { scope, caps: { day, week, month } } | null -- this job's scoped budget windows (issue #242,
136
- // INT-SCOPED-LIMITS-FILE-CONTRACT), resolved by the wiring from the same watched-limits snapshot the
137
- // pickup gate read. Null when the file is unset or the scope's row is concurrency-only; the default
138
- // keeps an unwired processor byte-identical. The folder MUTEX does not live here -- it is the pickup
139
- // gate's, pre-everything; this is only the money half.
140
- scopedCaps = null,
196
+ // [{ scope, keyPrefix, caps: { day, week, month }, reason }] -- this job's scoped job-count ledgers in reserve order
197
+ // (issues #242 and #499 part B, INT-SCOPED-LIMITS-FILE-CONTRACT): its repo or folder row's, then its project row's,
198
+ // from ONE builder (`scopedLedgers`) over the same watched-limits snapshot and pickup project the gate read. The
199
+ // global ledger is appended here, last. Empty when no row carries a job-count window; the default keeps an
200
+ // unwired processor byte-identical. The folder MUTEX and the `concurrent` slots do not live here -- they are the
201
+ // pickup gate's, pre-everything; this is only the money half.
202
+ scopedLedgers = [],
141
203
  recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
142
204
  // (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
143
205
  // this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
@@ -151,13 +213,21 @@ export async function runJob(job, deps) {
151
213
  // Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
152
214
  // unwired seam must not refuse, and the wiring is what turns the check on.
153
215
  checkWaitSkew = async () => ({ ok: true }),
154
- // () => { ok } | { message }. Issue #310. Resolves this deployment's provider credential the way
216
+ // () => { ok } | { message } | { unavailable } (issue #503: a transient overlay read, retried). Issue #310. Resolves this deployment's provider credential the way
155
217
  // buildContainerEnv will, and answers whether it exists AT ALL, so an unconfigured provider refuses
156
218
  // here rather than inside runContainer with the budget already reserved. Admit-everything by default,
157
219
  // like the two above and for their reason. A PROBE, deliberately: it discards whatever it resolves and
158
220
  // the real read happens where it always did, because threading a live credential through the processor
159
221
  // would put it in scope for every log line and record between here and the container.
160
222
  checkProviderCredential = () => ({ ok: true }),
223
+ // (refs) => { ok } | { unknown: { provider, id }, why } | { unavailable: code } (issue #502). Is each of this
224
+ // job's models, its main one and every listed one, a model pi knows (model-catalog.mjs `checkModelsKnown`,
225
+ // the builtin catalog plus the overlay models.json)? Admit-everything by default, like the credential probe
226
+ // above and for its reason: an unwired seam must not refuse, and the wiring is what turns the check on.
227
+ checkModelsKnown = () => ({ ok: true }),
228
+ // Issue #503: the pickup's endpoint snapshot `{ endpoints, models, set }` (index.mjs), read once per pickup. Null on
229
+ // a wiring with no endpoint seam, which leaves the credential gate and the container env exactly as before.
230
+ modelEndpoints = null,
161
231
  // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
162
232
  // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
163
233
  egressPreflight = async () => ({ ok: true }),
@@ -230,7 +300,7 @@ export async function runJob(job, deps) {
230
300
  mintToken,
231
301
  isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
232
302
  prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
233
- // runContainer({ job, token, prepared, secrets, name, signal, user, home }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
303
+ // runContainer({ job, token, prepared, secrets, name, signal, user, home, modelEndpoints? }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
234
304
  // `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
235
305
  // `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
236
306
  // `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
@@ -247,6 +317,49 @@ export async function runJob(job, deps) {
247
317
  // or a github job with no /outbox -- chains nothing. It NEVER throws (outbox.mjs), so its counts are
248
318
  // additive telemetry that can never flip the parent's completed outcome (CONST-RETRY-INFRA-ONLY).
249
319
  collectChain = async () => ({ enqueued: 0, refused: 0 }),
320
+ // The plan collector (issue #505, outbox-plan.mjs): a completed portfolio job's `/outbox/priorities.json`, handed to
321
+ // applyPlan. Called on the completed branch only, after collectChain, with the pickup's `portfolio` decision. It
322
+ // NEVER throws, so a refused plan is a recorded outcome of a completed job and never a retry. The default collects
323
+ // nothing (null: no plan), so a wiring that omits it records `plan: null`.
324
+ collectPlan = async () => null,
325
+ // Issue #501: the deployment's dollar windows `{ day, week, month }` in micro-dollars (each null when unset), or
326
+ // null when no window is set. Null is the default and the off switch: nothing is reserved or settled and no
327
+ // `budget:usd:*` key is written, so a deployment with no dollar setting is byte-identical.
328
+ dollarCaps = null,
329
+ // Issues #501 part 5 and #502 part 6 (scoped-limits.json version 2): this job's repo or folder dollar windows,
330
+ // `{ scope, keyPrefix, caps }` from `dollarCapsFor`, or null; and the model dollar windows it reserves in,
331
+ // `[{ ref, keyPrefix, caps }]` from `modelDollarRows` over its effective list (every model row when it has
332
+ // none). Both default to nothing, so an unwired processor reserves exactly what it did before.
333
+ scopedDollars = null,
334
+ // Issue #499 part B: this job's project row's dollar windows, `{ scope, keyPrefix, caps }` from
335
+ // `projectDollarCapsFor` with the project resolved at pickup, or null. Reserved after the repo or folder row's.
336
+ projectDollars = null,
337
+ modelDollars = [],
338
+ // Issue #504 part B (DES-DELEGATED-ALLOCATION-INSIDE-ENVELOPE): under an envelope, `governedDollars` narrows the
339
+ // ledgers above by the applied split, and these three ride beside them. `otherDollars` is `_other`'s ledger
340
+ // (`{ scope, keyPrefix, caps, capSource }`), for a job in no envelope project; `dollarCapSource` says, per window,
341
+ // whether the deployment cap came from the operator or the envelope total (`scopedDollars` and `projectDollars`
342
+ // carry their own `capSource`). `envelopeMismatch` is the pickup's verdict that this host's envelope is not the
343
+ // applied split's (true), or that it has none while one is applied ("no-envelope"). All default to nothing, so a deployment with no envelope is byte-identical.
344
+ otherDollars = null,
345
+ dollarCapSource = null,
346
+ envelopeMismatch = false,
347
+ // Issue #505: whether the LIVE triggers file still flags this job's cron trigger `run.portfolio: true`, as
348
+ // `(job) => boolean`, and this host's envelope `delegation` block (`{ enabled, writers }`), null with no envelope.
349
+ // The flag on the job data is what the trigger said when the job was queued; the file is what the operator says
350
+ // now, so removing the flag takes effect for a job already queued. The default answers "not flagged": an unwired
351
+ // processor treats every job as an ordinary one, which is what a job with no confirmed flag is.
352
+ checkPortfolioFlag = async () => false,
353
+ envelopeDelegation = null,
354
+ // Issue #504 part B: the project decided at pickup on the folder as named, and a function giving the project of a
355
+ // folder from the same projects snapshot. A local job whose RESOLVED folder belongs to another project is refused
356
+ // after prepare, before any reserve. Absent on a bare wiring, which checks nothing.
357
+ pickupProject = null,
358
+ folderProject = null,
359
+ // Issue #503 part 7: the builtin catalog's model object for (provider, id), or null (model-catalog.mjs
360
+ // `builtinModel`). Read only by the zero-rated check; the default knows no builtin model, so an unwired
361
+ // processor judges overlay models alone and reserves for every other.
362
+ builtinModel = () => null,
250
363
  now = new Date(),
251
364
  } = deps;
252
365
 
@@ -261,10 +374,29 @@ export async function runJob(job, deps) {
261
374
  const wantsForgeToken = isForgeBacked || job.github === true;
262
375
  let token = null;
263
376
  let prepared = null;
264
- let reserved = false;
265
- let scopedReserved = false;
377
+ // The job-count reservations still standing, in reserve order (issue #499 part B): what `reserveLedgers` took and
378
+ // no refund has given back yet. EVERY refund below is `releaseLedgers` over this one list, last first, so no path
379
+ // can give back one ledger and forget another, and none can give one back twice: a released ledger leaves the list.
380
+ const held = [];
381
+ // The global ledger's own entry, so `budgetReserved` can ask the list. ONE rule on every path (a count refusal, a
382
+ // dollar-cap, config-refused, an InfraRetry): `budgetReserved` says whether the GLOBAL slot is still held after any
383
+ // refund, global-only as INT-RUN-HISTORY-FILE-CONTRACT has it. A scoped or project slot a failed refund left behind
384
+ // is in the `budget_release_failed` log line, not in this field.
385
+ let globalLedger = null;
386
+ const globalHeld = () => globalLedger !== null && held.includes(globalLedger);
387
+ // Give back every held job-count slot, NEVER throwing: a refund that did not land must not replace the caller's
388
+ // classification with a Redis message. Logged with `at`; returns whether the list is empty afterwards.
389
+ const refundLedgers = async (at) => {
390
+ try {
391
+ await releaseLedgers(redis, held, { now });
392
+ return true;
393
+ } catch (releaseError) {
394
+ log("budget_release_failed", { at, code: releaseError?.code ?? null });
395
+ return false;
396
+ }
397
+ };
266
398
  // Set once `runContainer` has RESOLVED, which is the only moment a container is known to have run. The
267
- // config classifier in the catch refunds both ledgers, and its whole justification is that nothing was
399
+ // config classifier in the catch refunds every job-count ledger, and its whole justification is that nothing was
268
400
  // spent; without a fact to test, that is a claim about where config throws happen to live today rather
269
401
  // than a property of the code. A config-tagged throw raised after a paid run would otherwise refund a slot
270
402
  // the container really spent AND tell the operator publicly that nothing was.
@@ -275,6 +407,11 @@ export async function runJob(job, deps) {
275
407
  // A spawn fault that fails between is an InfraRetry carrying `container-never-started`, refunded by the
276
408
  // arm below, which has a discriminator of its own.
277
409
  let containerRan = false;
410
+ // Issue #501. `dollarHold` is the dollar reservation still standing (reserveDollars' hold), null when none was
411
+ // taken or once it has been settled or given back, so no path can settle or refund it twice. `dollars` is what
412
+ // the record says about it (INT-RUN-HISTORY-FILE-CONTRACT), null until the reservation step ran.
413
+ let dollarHold = null;
414
+ let dollars = null;
278
415
 
279
416
  try {
280
417
  // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
@@ -306,6 +443,18 @@ export async function runJob(job, deps) {
306
443
  // standing between a stale receiver and a paid job that ran when the operator wrote "wait".
307
444
  {
308
445
  const skew = await checkWaitSkew(job);
446
+ // Issue #502: an authored NARROWING field the job arrived without (`AUTHORED_NARROWING_FIELDS`, triggers-file.mjs),
447
+ // today `run.models` and `run.maxCostUsd`. The same two causes as a dropped wait, and the same refusal shape; without it the job would
448
+ // run on the deployment's list, or on none, while every record reads like a correct run. The FIELD is named,
449
+ // never its value: the comment's reader may be an issue author.
450
+ if (skew.skewed && typeof skew.field === "string") {
451
+ // Three causes, the likeliest first (PR #536's review, round 3, made the check strict): the trigger gained the
452
+ // field after this job was queued, a service is below the version that carries it, or one still reads an
453
+ // older copy of the triggers file. The first is fixed by re-running the job, so the comment says so.
454
+ await comment(job, `Refused: this trigger sets \`run.${skew.field}\`, but the job reached the worker without it, so it would have run without that limit. Either the trigger changed after this job was queued (re-run it), or a service in this deployment is stale: below the version that carries the field, or still reading an older copy of the triggers file and in need of a restart. Not run.`);
455
+ log("refused_trigger_skew", { triggerIndex: job.trigger?.matched?.index ?? null, field: skew.field, causes: "trigger-changed-after-queue-or-stale-service" });
456
+ return { outcome: "policy", reason: "trigger-skew", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
457
+ }
309
458
  if (skew.skewed) {
310
459
  // Named for the operator, not the payload: how many conditions were authored, never what
311
460
  // they say. The fix is a version, so the comment says which one.
@@ -487,57 +636,16 @@ export async function runJob(job, deps) {
487
636
  log("refused_image_forge_unsupported", { image: img.forgeUnsupported, kind: img.kind, declared: img.declared });
488
637
  return { outcome: "policy", reason: "job-image-forge-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
489
638
  }
490
- if (img.replicaUnsupported) {
491
- // The image is present and does not declare replica support (REQ-REPLICA-RUNS), so its baked
492
- // HARD_RULES.md predates the amendment and still hard-codes `pi/issue-<n>` as a SYSTEM rule --
493
- // which the model treats as authoritative over the user prompt naming `pi/issue-<n>-r2`. Both
494
- // replicas would push to one branch: not an error, just the push race the feature exists to
495
- // avoid, with two runs billed and one pull request to show for it.
496
- //
497
- // Determinate, so a refusal rather than a retry, and pre-spend, because no version of this gets
498
- // better by running. Like the forge branch above, the message names the FIX rather than the label
499
- // that noticed it -- an operator reading "rebuild the image" is already where they need to be.
500
- await comment(
501
- job,
502
- `Refused: the job image "${img.replicaUnsupported}" does not declare replica support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its baked guardrails would name the wrong branch. Rebuild the image from a version that has this feature. Not run.`,
503
- );
504
- log("refused_image_replicas_unsupported", { image: img.replicaUnsupported, declared: img.declared });
505
- return { outcome: "policy", reason: "job-image-replicas-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
506
- }
507
- if (img.commandUnsupported) {
508
- // The image is present and does not declare command support (issue #189), so its runner
509
- // predates run.command: it reads no PI_COMMAND, and the bare `/name args` prompt reaches the
510
- // model as PROSE -- no handler runs, the agent improvises, and the queue records a clean exit
511
- // 0. The in-container half of the gate (the runner's own command-unregistered refusal) does
512
- // not exist on such an image, which is exactly why the host must refuse first.
513
- //
514
- // Determinate, so a refusal rather than a retry, and pre-spend, because no version of this
515
- // gets better by running. Like the replica branch above, the message names the FIX rather
516
- // than the label that noticed it.
517
- await comment(
518
- job,
519
- `Refused: the job image "${img.commandUnsupported}" does not declare command support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its runner would not dispatch \`run.command\`. Rebuild the image from a version that has this feature. Not run.`,
520
- );
521
- log("refused_image_commands_unsupported", { image: img.commandUnsupported, declared: img.declared });
522
- return { outcome: "policy", reason: "job-image-commands-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
523
- }
524
- if (img.excludeToolsUnsupported) {
525
- // The image is present and does not declare exclude-tools support (issue #291), so its runner
526
- // predates run.excludeTools: it reads no PI_EXCLUDE_TOOLS, and the job would run with every
527
- // tool the trigger says to remove -- a "read-only" trigger with a working editor and shell,
528
- // recording a clean exit. That is a PERMISSION quietly not enforced, the silent fail-open this
529
- // repo brands the worst outcome available, which is exactly why the host refuses before spend
530
- // rather than letting the container fail open.
531
- //
532
- // Determinate, so a refusal rather than a retry, and pre-spend, because no version of this
533
- // gets better by running. Like the command branch above, the message names the FIX rather
534
- // than the label that noticed it.
535
- await comment(
536
- job,
537
- `Refused: the job image "${img.excludeToolsUnsupported}" does not declare exclude-tools support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its runner would ignore \`run.excludeTools\` and run this trigger with every tool it says to remove. Rebuild the image from a version that has this feature. Not run.`,
538
- );
539
- log("refused_image_exclude_tools_unsupported", { image: img.excludeToolsUnsupported, declared: img.declared });
540
- return { outcome: "policy", reason: "job-image-exclude-tools-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
639
+ // The image capability gates (one table, CAPABILITY_GATES in image-preflight.mjs): the image is present and
640
+ // does not declare a feature this job carries, so its runner would silently ignore it. Determinate, so a
641
+ // refusal rather than a retry, and pre-spend, because no version of this gets better by running. Like the
642
+ // forge branch above, each row's comment names the FIX rather than the label that noticed it.
643
+ const gate = CAPABILITY_GATES.find((row) => img[row.result]);
644
+ if (gate) {
645
+ const image = img[gate.result];
646
+ await comment(job, gate.comment(image, img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"));
647
+ log(gate.event, { image, declared: img.declared });
648
+ return { outcome: "policy", reason: gate.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
541
649
  }
542
650
  if (img.unavailable) {
543
651
  // docker itself did not answer -- transient infra, NOT a determinate refusal. THROWN so BullMQ
@@ -568,6 +676,65 @@ export async function runJob(job, deps) {
568
676
  throw new InfraRetry("the job user could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
569
677
  }
570
678
 
679
+ // Issue #502, two free gates on the job's models, BEFORE the credential gate so a typo in a model or provider
680
+ // id is named as itself rather than as a missing key, and so before the egress probe, the secret resolvers,
681
+ // the mint, the clone, the token-cap read and both reserves (CONST-BUDGET-BEFORE-TOKENS). Both read the
682
+ // EFFECTIVE job (index.mjs `effectiveJobOf`): the main model after the `job.data > overlay > env` fill, and
683
+ // the list as `job.data.models ?? PI_ALLOWED_MODELS`. Checking `job.data` alone would wave through a model
684
+ // the overlay or the env supplied, which is the common case: most forge triggers name none.
685
+ //
686
+ // The model ids are operator configuration, so the log names them; the forge comment does not, for the
687
+ // terminal comments' reason: its reader may be an issue author, who can act on neither.
688
+ {
689
+ // A list that is PRESENT but not a valid list (a string, `{}`, `0`, `false`, `""`, a bad entry) refuses here
690
+ // rather than reading as "no list": the loader and the env parser both refuse such a value, so it is a
691
+ // hand-built or foreign job, and treating it as absent would let it run unrestricted (or hide the
692
+ // deployment's own list behind it). Same refusal as an unknown model, with its own `why`.
693
+ if (job.models !== undefined && job.models !== null && modelListProblem(job.models) !== null) {
694
+ await comment(job, MODEL_UNKNOWN_COMMENT);
695
+ log("refused_model_unknown", { provider: job.provider ?? null, model: job.model ?? null, why: "list-malformed" });
696
+ return { outcome: "policy", reason: "model-unknown", why: "list-malformed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
697
+ }
698
+ const refs = [{ provider: job.provider, id: job.model, main: true }];
699
+ for (const entry of Array.isArray(job.models) ? job.models : []) {
700
+ const ref = splitModelEntry(entry);
701
+ refs.push({ provider: ref.provider, id: ref.model });
702
+ }
703
+ const known = await checkModelsKnown(refs);
704
+ if (known?.unavailable) {
705
+ // The overlay models.json could not be READ just now (an errno, never file text). Retried, never
706
+ // refused, the credential gate's rule for the same file below.
707
+ log("model_catalog_unavailable", { provider: job.provider ?? null, model: job.model ?? null, reason: known.unavailable });
708
+ throw new InfraRetry("whether this job's models exist could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
709
+ }
710
+ if (known?.unknown) {
711
+ // An `overlay-*` why is the deployment's file, not the job's model: its own comment.
712
+ const why = typeof known.why === "string" ? known.why : null;
713
+ await comment(job, why?.startsWith("overlay-") ? OVERLAY_REFUSED_COMMENT : MODEL_UNKNOWN_COMMENT);
714
+ // `why` is a fixed token: `overlay-unparseable` tells the operator the file is the problem, not the id.
715
+ log("refused_model_unknown", { provider: known.unknown.provider ?? null, model: known.unknown.id ?? null, why: known.why ?? null });
716
+ return { outcome: "policy", reason: "model-unknown", why, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
717
+ }
718
+ // The main model must be on the job's list. The loader refuses this when one trigger names all three, so
719
+ // what reaches here is a model or provider the overlay or the env supplied: a `PI_ALLOWED_MODELS` that
720
+ // does not list `PI_MODEL`, or a `dispatch_set model` that moved the default off a trigger's list. Both
721
+ // halves are compared, exact (`modelOnList`), the runner guard's rule.
722
+ if (Array.isArray(job.models) && !modelOnList(job.models, job.provider, job.model)) {
723
+ await comment(job, "Refused: the AI model this job would run on is not on the list of models it is allowed to use, so no container was started and nothing was spent. Ask the operator to check the trigger's model settings. Not run.");
724
+ log("refused_model_not_allowed", { provider: job.provider ?? null, model: job.model ?? null });
725
+ return { outcome: "policy", reason: "model-not-allowed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
726
+ }
727
+ // A listed model whose declared fallbacks are not all listed (model-catalog.mjs `declaredFallbacks`): on
728
+ // anthropic-messages pi sends them with every call, so the runner's guard would refuse every call to it after
729
+ // the container started; on any other api the worker is stricter than the runner, by decision. Refused here, free. The log names the listed model, the operator's own configuration; the
730
+ // comment names none.
731
+ if (known?.fallbackUnlisted) {
732
+ await comment(job, "Refused: a model this job is allowed to use declares fallback models that are not on the job's list, so no container was started and nothing was spent. Ask the operator to list those models too, or remove that model. Not run.");
733
+ log("refused_model_not_allowed", { provider: known.fallbackUnlisted.provider ?? null, model: known.fallbackUnlisted.id ?? null, why: "fallback-unlisted" });
734
+ return { outcome: "policy", reason: "model-not-allowed", why: "fallback-unlisted", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
735
+ }
736
+ }
737
+
571
738
  // Is there a credential to run this job with at all? FREE, determinate and I/O-light: a pure function
572
739
  // of the job's provider, the worker env, and (only when the env has no key) one small readFileSync of
573
740
  // pi's auth.json. So it goes here, with the other free gates, which is further than issue #310 asked
@@ -590,7 +757,61 @@ export async function runJob(job, deps) {
590
757
  // The refusal names no path. `credentialFromPiAuth`'s messages carry auth.json's location, `comment`
591
758
  // posts publicly on the issue, and the reason an operator needs is the same either way: their
592
759
  // deployment has no usable provider credential and `doctor` will say exactly which variable.
593
- const credential = await checkProviderCredential(job);
760
+ // `modelEndpoints` is the pickup's snapshot (issue #503), handed to the gate and to runContainer alike, so a
761
+ // keyless provider passes here and gets its PI_DISPATCH_KEYLESS there from ONE read of the declaration. runContainer
762
+ // is handed it only when an endpoint is declared, so with none its context is byte-identical to before.
763
+ // THE ENVELOPE GATE (issue #504 part B, DES-DELEGATED-ALLOCATION-INSIDE-ENVELOPE): this host's envelope digest is not
764
+ // the applied split's, so the split this job would be judged against was computed for another envelope. FREE and
765
+ // determinate (the pickup read decided it), so it sits with the free gates, before the mint, the clone, the
766
+ // token-cap read and every reserve (CONST-BUDGET-BEFORE-TOKENS), and it RETURNS: a retry meets the same envelope
767
+ // until the operator makes the hosts agree (CONST-RETRY-INFRA-ONLY). Refusing is loud and money-safe; judging one
768
+ // split against two envelopes is neither.
769
+ // `envelopeMismatch` is true (this host's envelope is another one) or "no-envelope" (it has none while the fleet has
770
+ // an applied split): one reason, two texts, since the fix differs.
771
+ if (envelopeMismatch === true || envelopeMismatch === "no-envelope") {
772
+ const absent = envelopeMismatch === "no-envelope";
773
+ await comment(
774
+ job,
775
+ absent
776
+ ? "Refused: this worker has no budget envelope while the other workers share an applied budget split, so no container was started and nothing was spent. Ask the operator to install the envelope on this worker, or to turn delegated allocation off for every worker (`pi-dispatch doctor` says how). Not run."
777
+ : "Refused: this worker's budget envelope differs from the one the current budget split was made for, so no container was started and nothing was spent. Ask the operator to run `pi-dispatch doctor`. Not run.",
778
+ );
779
+ log("refused_envelope_mismatch", absent ? { envelope: "none" } : {});
780
+ return { outcome: "policy", reason: ENVELOPE_MISMATCH_REASON, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
781
+ }
782
+
783
+ // THE PORTFOLIO GATE (issue #505, `portfolio-no-envelope`). A portfolio job exists to write a priorities plan, and a
784
+ // plan applies only on a host whose envelope has delegation on with `portfolio-job` among its writers. Without
785
+ // that the plan is refused after the job has paid to write it, so the job is refused here instead: FREE (the
786
+ // envelope was read at boot or reload, the triggers file is one local read), before the mint, the clone, the
787
+ // token-cap read and every reserve (CONST-BUDGET-BEFORE-TOKENS), and RETURNED, since a retry meets the same
788
+ // envelope (CONST-RETRY-INFRA-ONLY). After the envelope gate, which names the host-wide fault first.
789
+ //
790
+ // The live flag comes FIRST. The job data says what the trigger said when it was queued; the live file says what
791
+ // the operator says now. A job whose trigger no longer flags it (or whose file cannot be read) is an ordinary
792
+ // cron job: it runs unflagged and passes this gate, rather than being refused for a flag nobody holds any more.
793
+ // Only a job with a cron `trigger` and no chain fields is asked about at all: a manual run and a chained child can
794
+ // never be a portfolio job, whatever their data says.
795
+ const portfolio = job.portfolio === true && job.trigger !== undefined && job.parentJobId === undefined && job.chainDepth === undefined && (await Promise.resolve().then(() => checkPortfolioFlag(job)).catch(() => false)) === true;
796
+ if (portfolio) {
797
+ const delegation = envelopeDelegation;
798
+ const why = delegation === null || delegation === undefined ? "no-envelope" : delegation.enabled !== true ? "delegation-off" : !Array.isArray(delegation.writers) || !delegation.writers.includes("portfolio-job") ? "writer-not-allowed" : null;
799
+ if (why !== null) {
800
+ // No comment: a portfolio job is always local, and a local job has no issue to comment on. `why` is a
801
+ // fixed token, so the log says which of the three the operator has to change.
802
+ log("refused_portfolio_no_envelope", { why });
803
+ return { outcome: "policy", reason: PORTFOLIO_NO_ENVELOPE, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
804
+ }
805
+ }
806
+
807
+ const credential = await checkProviderCredential(job, { modelEndpoints });
808
+ if (credential.unavailable) {
809
+ // Issue #503: the overlay models.json could not be read at this pickup for a transient reason, so the gate has no
810
+ // verdict for a provider that may be keyless. Retried, never refused: a refusal is permanent and public, and the
811
+ // next attempt may read the file. The code is a fixed errno token, never file text.
812
+ log("provider_credential_unavailable", { provider: job.provider ?? null, reason: credential.unavailable });
813
+ throw new InfraRetry("the provider credential could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
814
+ }
594
815
  if (!credential.ok) {
595
816
  // The PROVIDER is named and the message is not. The provider is operator-authored config, already
596
817
  // on the record and the mirror, and named freely by the sibling refusals (`backend-unblessed` names
@@ -732,7 +953,16 @@ export async function runJob(job, deps) {
732
953
  //
733
954
  // Guarded by `secretsArmed` at the CALL SITE, not inside the resolver: an unflagged job must not reach
734
955
  // it under ANY wiring, and a guard that lives in the default is a guard an injected resolver skips.
735
- const resolved = secretsArmed(job) ? await resolveSecrets(job) : { ok: true, secrets: {} };
956
+ //
957
+ // The load-time reserved names are asked again HERE, of the job itself, for the same reason (issue
958
+ // #511). parseTriggers refuses them when the file loads, but a job does not always come from a file
959
+ // this worker loaded: one queued before an upgrade widened the set, a cron job-scheduler template
960
+ // stored in Valkey (schedules.mjs keeps `run.secrets` in it), or a receiver older than the worker all
961
+ // carry `secrets` the current set would refuse, and buildContainerEnv would write every one of them.
962
+ // The same set the loader uses, imported, so the two cannot drift, and checked before the resolver so
963
+ // no wiring of it can skip the check.
964
+ const loadReserved = secretsArmed(job) ? Object.keys(job.secrets ?? {}).find((name) => RESERVED_ENV_NAMES.has(name)) : undefined;
965
+ const resolved = loadReserved !== undefined ? { reserved: loadReserved, atLoad: true } : secretsArmed(job) ? await resolveSecrets(job) : { ok: true, secrets: {} };
736
966
  if (resolved.profileUnknown) {
737
967
  await comment(job, "Refused: this trigger set `run.secrets`, and the resolver profile it names is not usable on this worker host. No profile of that name is declared, or its resolver is absent or not executable. The job would have started with those variables unset, and an agent that gets a 401 writes a plausible report and exits 0. Run `pi-dispatch doctor` on the worker to see which profiles it has. Not run.");
738
968
  // The operator's own profile LABEL, and never a path, a reference, or a byte the resolver printed.
@@ -771,7 +1001,14 @@ export async function runJob(job, deps) {
771
1001
  // (apiKeyVariable skips both), so a trigger binding one lands beside the operator's key and
772
1002
  // outranks it in pi's own precedence.
773
1003
  // The old message asserted the first for both, which is exactly backwards for the second.
774
- await comment(job, `Refused: this trigger's \`run.secrets\` binds \`${resolved.reserved}\`, which is a variable this deployment already uses for the job's own credentials. Whichever of the two values reached the container, one of them would be silently ignored. Rename it in the triggers file. Not run.`);
1004
+ // A name the triggers file itself would refuse at load (issue #511) reaches here only from a job
1005
+ // queued before that refusal, a stored cron scheduler template, or an older receiver, so it says that.
1006
+ await comment(
1007
+ job,
1008
+ resolved.atLoad
1009
+ ? `Refused: this job's \`run.secrets\` binds \`${resolved.reserved}\`, a name the triggers file refuses at load because this deployment writes it or pi reads it to configure a provider or itself. The job was queued before that refusal applied, by a stored cron schedule, or by an older receiver. Rename it in the triggers file. Not run.`
1010
+ : `Refused: this trigger's \`run.secrets\` binds \`${resolved.reserved}\`, which is a variable this deployment already uses for the job's own credentials. Whichever of the two values reached the container, one of them would be silently ignored. Rename it in the triggers file. Not run.`,
1011
+ );
775
1012
  // The variable NAME only. It is the operator's own choice of name, not payload, and naming it is what
776
1013
  // makes the refusal actionable -- but the REFERENCE behind it never appears.
777
1014
  log("refused_secret_name_reserved", { kind: job.kind ?? null, name: resolved.reserved });
@@ -797,7 +1034,7 @@ export async function runJob(job, deps) {
797
1034
  // than by matching its stderr, which image-preflight.mjs forbids for good reason. Folding this into
798
1035
  // the refusal above would permanently burn a delivery over a twenty-second vault blip, and a webhook
799
1036
  // does not redeliver itself. Nothing has spent: `budgetReserved` computes false in the catch below
800
- // because `reserved` is still false here.
1037
+ // because no ledger is held yet.
801
1038
  log("secret_resolver_unreachable", { kind: job.kind ?? null, name: resolved.unreachable, failure: resolved.failure ?? null, code: resolved.code ?? null, stderrBytes: resolved.stderrBytes ?? 0 });
802
1039
  throw new InfraRetry(`secret resolver could not answer for ${resolved.unreachable}`, { reason: "secret-resolver-unreachable", provider: job.provider ?? null, model: job.model ?? null });
803
1040
  }
@@ -828,16 +1065,35 @@ export async function runJob(job, deps) {
828
1065
  }
829
1066
 
830
1067
  // `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
831
- prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
1068
+ // `portfolio` (issue #505) only for a job the gate above confirmed: prepare then writes /job/portfolio.json, after
1069
+ // asking the live file once more. Absent otherwise, so every other job's prepare call is unchanged.
1070
+ prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}), ...(portfolio ? { portfolio: true } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
832
1071
 
833
1072
  // A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
834
1073
  // or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
835
1074
  // -- is POLICY: return before reserveBudget so it burns no cap slot and is never retried.
836
1075
  // Mirrors the branch-protection policy return above. Spread-plus-attribution: the prepare
837
1076
  // result keeps its own reason and fields, and the host-effective provider/model land beside
838
- // them exactly as on every other terminal result.
1077
+ // them exactly as on every other terminal result. `budgetReserved: false` like every pre-reserve refusal (issue
1078
+ // #507): nothing was reserved and no container started, so the record says so and the cost fold counts the run as
1079
+ // an exact $0 rather than a floor (REQ-COST-ANALYTICS (d)). After the spread, so no preparer can say otherwise.
839
1080
  if (prepared?.outcome === "policy") {
840
- return { ...prepared, provider: job.provider ?? null, model: job.model ?? null };
1081
+ return { ...prepared, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false };
1082
+ }
1083
+
1084
+ // THE RESOLVED FOLDER'S PROJECT (issue #504 part B). The pickup decided the project, and so the share a job reserves
1085
+ // against, from the folder AS NAMED; the container mounts the folder prepare RESOLVED. A link or a case variant
1086
+ // inside a run root that leads into ANOTHER project's member would bill that work to the named project's share.
1087
+ // Refused here, after prepare and before the token-cap read and every reserve: free, determinate, never retried. A
1088
+ // resolved folder in no project is not refused, because the named spelling is the operator's own membership
1089
+ // (a symlinked member is listed by the path the triggers use, docs/projects.md).
1090
+ if (job.kind === "local" && typeof folderProject === "function" && typeof prepared?.workspace === "string") {
1091
+ const resolvedProject = folderProject(prepared.workspace);
1092
+ if (resolvedProject !== null && resolvedProject !== pickupProject) {
1093
+ await comment(job, "Refused: this job's folder now leads into another project's folder, so no container was started and nothing was spent. Not run.");
1094
+ log("refused_local_folder_project_changed", { pickup: pickupProject, resolved: resolvedProject });
1095
+ return { outcome: "policy", reason: LOCAL_FOLDER_PROJECT_CHANGED, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
1096
+ }
841
1097
  }
842
1098
 
843
1099
  // Daily TOKEN cap (issue #25): the deliberate check-AFTER control. Token cost is only known
@@ -857,75 +1113,169 @@ export async function runJob(job, deps) {
857
1113
  return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
858
1114
  }
859
1115
 
860
- // Per-scope budget windows (issue #242, INT-SCOPED-LIMITS-FILE-CONTRACT): the NARROWER ledger
861
- // reserves FIRST, so a noisy scope's refusals never consume a global slot -- the global INCR below
862
- // runs only for jobs the scope admitted. Same atomic INCR, same refused-still-counts invariant,
863
- // through budget.mjs's keyPrefix seam (dayKey/weekKey/monthKey under budget:s:<hash16>). softHoldPct
864
- // is deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob;
865
- // scoped windows are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
866
- if (scopedCaps) {
867
- // A redis fault BETWEEN this reserve and the global one below strands the scoped INCR with no
868
- // run and no refund -- the pre-existing mid-reserve posture, shared with the global ledger's
869
- // own partial-INCR seam; the compensating release below covers REFUSALS, not faults.
870
- const scoped = await reserveBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
871
- scopedReserved = true;
872
- if (!scoped.allowed) {
873
- const w = scoped.blockedWindow;
874
- const win = scoped.windows[w];
875
- // A local job's scope is a full host path and its "comment" is not dropped -- the wiring's
876
- // local adapter LOGS the text (start.mjs forgeFor fallthrough) -- so the path must never
877
- // enter the message; "this folder" is enough beside the jobId the adapter logs. A forge
878
- // scope IS the repo the comment posts on, safe to name.
879
- const scopeLabel = job.kind === "local" ? "this folder" : scopedCaps.scope;
880
- await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
881
- // The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would
882
- // put a full host path in the worker log against no-pii-in-logs (the record keeps only
883
- // basename(folder) for the same reason). The admin recomputes the key from the configured
884
- // scope to join it back.
885
- log("over_scope_budget", { scopeKey: scopeKeyPrefix(scopedCaps.scope), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
886
- // budgetReserved false: the GLOBAL slot was never touched (scoped reserves first). The scoped
887
- // counter did INCR and keeps it -- its own refused-reservation-still-counts, per ledger.
888
- return { outcome: "policy", reason: "scope-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
1116
+ // The JOB-COUNT ledgers (issues #242 and #499 part B, INT-SCOPED-LIMITS-FILE-CONTRACT), narrowest first: the repo
1117
+ // or folder row, the project row, then the global windows, reserved in that order by ONE helper. A narrow
1118
+ // ledger's refusal never consumes a slot in a wider one -- the global INCR runs only for jobs every scoped ledger
1119
+ // admitted, and a project's only for jobs their repo admitted. Same atomic INCR, same refused-still-counts
1120
+ // invariant per ledger, through budget.mjs's keyPrefix seam (budget:s:<hash16> of the row scope). softHoldPct is
1121
+ // deliberately GLOBAL-ONLY: the band is one operator brake on overall spend, not a per-row knob; scoped windows
1122
+ // are hard caps (DES-SCOPED-LIMITS-AND-FOLDER-MUTEX).
1123
+ //
1124
+ // A redis fault mid-walk gives back every LEDGER that landed whole (`held` is exact) and rethrows. Inside the
1125
+ // ledger that faulted, the windows INCRed before the fault stay counted (a week INCR that faults leaves that
1126
+ // ledger's day counted, an EXPIRE that faults leaves its key without a TTL): `reserveBudget` does not say which of
1127
+ // its windows landed, so giving them back could DECR a window that never rose. That is the pre-existing
1128
+ // mid-reserve posture of one ledger, unchanged.
1129
+ globalLedger = { scope: null, keyPrefix: null, caps, softHoldPct, reason: null };
1130
+ let counted;
1131
+ try {
1132
+ counted = await reserveLedgers(redis, [...(scopedLedgers ?? []), globalLedger], held, { now });
1133
+ } catch (error) {
1134
+ // Valkey failed mid-walk. No container can have started, and `held` lists exactly the reservations that
1135
+ // landed, so they go back (last first, never throwing) before the error escapes; otherwise a repo and a
1136
+ // project slot would stay counted for a job that never ran.
1137
+ await refundLedgers("reserve-fault");
1138
+ throw error;
1139
+ }
1140
+ if (!counted.allowed) {
1141
+ // Every ledger BEFORE the refusing one gives its slot back, last first: a ledger that did not issue the
1142
+ // refusal gives back. Without this, an exhausted global window drains every arriving scope's and project's own
1143
+ // counters with zero runs to show for it, and a full project drains its members' repo windows. The refusing
1144
+ // ledger keeps its own slot (refused-still-counts, per ledger).
1145
+ await refundLedgers(counted.refusedBy.reason ?? counted.result.reason);
1146
+ const result = counted.result;
1147
+ const w = result.blockedWindow;
1148
+ const win = result.windows[w];
1149
+ if (counted.refusedBy === globalLedger) {
1150
+ if (result.reason === "soft-hold") {
1151
+ await comment(job, `Soft-hold: ${w} spend ${win.reserved}/${win.cap} is inside the ${softHoldPct}% hold band. New starts paused; not run.`);
1152
+ log("soft_hold", { window: w, reserved: win.reserved, cap: win.cap, pct: softHoldPct });
1153
+ } else {
1154
+ await comment(job, `Over the ${w} budget cap (${win.cap}). Not run.`);
1155
+ log("over_budget", { window: w, reserved: win.reserved, cap: win.cap });
1156
+ }
1157
+ // budgetReserved true: the global slot is reserved and kept (a refused reservation still counts). Both
1158
+ // over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
1159
+ return { outcome: "policy", reason: result.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
889
1160
  }
1161
+ const isProject = counted.refusedBy.reason === PROJECT_CAP_REASON;
1162
+ // A local job's scope is a full host path and its "comment" is not dropped -- the wiring's local adapter LOGS
1163
+ // the text (start.mjs forgeFor fallthrough) -- so the path must never enter the message; "this folder" is enough
1164
+ // beside the jobId the adapter logs. A forge scope IS the repo the comment posts on, safe to name, and named
1165
+ // without a forge prefix (issue #498). A project is "this project": the comment's reader may be an issue
1166
+ // author, and which repos an operator groups is the operator's business, not theirs.
1167
+ const scopeLabel = isProject ? "this project" : job.kind === "local" ? "this folder" : unqualifiedScope(counted.refusedBy.scope);
1168
+ await comment(job, `Over the ${w} run cap for ${scopeLabel} (${win.cap}). Not run.`);
1169
+ // The scope rides the log as its 16-hex key, NEVER the raw string: a folder-scoped cap would put a full host
1170
+ // path in the worker log against no-pii-in-logs. The admin recomputes the key from the configured scope. A
1171
+ // project refusal adds `ledger: "project"`; a repo or folder one keeps the line it always had.
1172
+ log("over_scope_budget", { scopeKey: counted.refusedBy.keyPrefix, ...(isProject ? { ledger: "project" } : {}), window: w, reserved: win.reserved, cap: win.cap, kind: job.kind === "local" ? "local" : "forge" });
1173
+ // budgetReserved false: the GLOBAL slot was never touched (every scoped ledger reserves first).
1174
+ return { outcome: "policy", reason: counted.refusedBy.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
890
1175
  }
891
1176
 
892
- // GLOBAL budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
893
- // every active window (day + optional week/month) and the soft-hold band in one atomic pass.
894
- const budget = await reserveBudget(redis, { caps, softHoldPct, now });
895
- reserved = true;
896
- if (!budget.allowed) {
897
- // The scoped reserve above committed before this global refusal -- give that slot back. Without
898
- // this, an exhausted global window drains every arriving scope's own day/week/month counters
899
- // with zero runs to show for it (a storm against a spent global daily cap would empty a repo's
900
- // week by noon). The scoped ledger's refused-still-counts covers the SCOPE's own refusal above,
901
- // never a refusal it did not issue.
902
- if (scopedReserved && scopedCaps) {
903
- await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
904
- // CLEARED, so the ledger's state and the flag agree. Nothing between here and the return can
905
- // throw today (`comment` is non-throwing by construction and `log` is a write), but the catch
906
- // below now releases on a whole CLASS of error rather than one reason, and a second release
907
- // against this same key would take it to -1: `releaseBudget` is a floorless DECR. The invariant
908
- // belongs where the release is, not in the guard of every future reader.
909
- scopedReserved = false;
910
- }
911
- const w = budget.blockedWindow;
912
- const win = budget.windows[w];
913
- if (budget.reason === "soft-hold") {
914
- await comment(job, `Soft-hold: ${w} spend ${win.reserved}/${win.cap} is inside the ${softHoldPct}% hold band. New starts paused; not run.`);
915
- log("soft_hold", { window: w, reserved: win.reserved, cap: win.cap, pct: softHoldPct });
1177
+ // DOLLAR windows (issue #501, part 3), after EVERY job-count reserve and before the container: the last gate
1178
+ // before the spend line (CONST-BUDGET-BEFORE-TOKENS). Every free gate and every job-count ledger come first, so
1179
+ // a refusal anywhere above never touches a dollar counter. The amount is the job's per-job cost cap, which the
1180
+ // runner enforces before every call, so it bounds what the run can spend; the worker needs no prices.
1181
+ let containerJob = job;
1182
+ // Every dollar ledger this job reserves in: the deployment's windows, its repo or folder row's, and each model
1183
+ // row it may reach (scoped-limits.json version 2). ONE reservation over all of them, so a refusal in any window
1184
+ // gives back every key, the deployment's included.
1185
+ const ledgers = dollarLedgers(dollarCaps, { scope: scopedDollars, project: projectDollars, other: otherDollars, models: modelDollars });
1186
+ const modelPrefixes = new Map((modelDollars ?? []).map((m) => [m.keyPrefix, m.ref]));
1187
+ // The ledgers that settle to the JOB's cost (the deployment's, the repo or folder row's, the project row's), named
1188
+ // rather than derived as "not a model", so a ledger kind added later settles nowhere until it is put on a list.
1189
+ // `_other`'s ledger (issue #504 part B) settles to the job's cost too: its counter is what the unassigned work spent.
1190
+ const jobCostPrefixes = new Set([DOLLAR_KEY_PREFIX, scopedDollars?.keyPrefix, projectDollars?.keyPrefix, otherDollars?.keyPrefix].filter((p) => typeof p === "string"));
1191
+ if (ledgers.length > 0) {
1192
+ // A window needs a per-job cap (the settings invariant, `checkDollarInvariant`; for a scoped or model row, the
1193
+ // same rule), so a job with none here is a defect, a hand-built queue entry, or a dollar row in
1194
+ // scoped-limits.json on a deployment with no `maxCostUsd`: refused as configuration, before anything is
1195
+ // reserved, and the config arm below refunds every job-count slot.
1196
+ if (job.maxCostMicros === null || job.maxCostMicros === undefined) throw configError("a dollar window is set but this job has no per-job cost cap to reserve");
1197
+ // Issue #503 part 7: a job that CANNOT spend reserves nothing. Every model it may call is served by a declared
1198
+ // endpoint and zero-rated, so its container runs under a per-job cap of 0, which the runner's cost guard holds
1199
+ // before every call: a call whose bound is above 0 is refused, so the job's spend is 0 by construction, not
1200
+ // by trust in the cost table. The capability gate above already required `costCap` of this job's image (the
1201
+ // job carried a non-null cap), so an image that would ignore the 0 never gets here. A cap that is ALREADY 0
1202
+ // (a malformed queued value reads as 0, `effectiveCostCapMicros`) has nothing to reserve either.
1203
+ const refs = callableModelRefs(job);
1204
+ const zero = job.maxCostMicros === 0 ? { zeroRated: true } : zeroRatedVerdict({ models: modelEndpoints?.models ?? null, endpoints: modelEndpoints?.endpoints ?? [], refs, builtinModel });
1205
+ if (zero.zeroRated) {
1206
+ containerJob = { ...job, maxCostMicros: 0 };
1207
+ dollars = dollarsRecord({ reservedMicros: 0, settledMicros: 0, basis: "unreserved" });
1208
+ log("dollar_unreserved", { models: refs.length });
916
1209
  } else {
917
- await comment(job, `Over the ${w} budget cap (${win.cap}). Not run.`);
918
- log("over_budget", { window: w, reserved: win.reserved, cap: win.cap });
1210
+ let reservation;
1211
+ try {
1212
+ reservation = await reserveDollars(redis, { ledgers, amountMicros: job.maxCostMicros, now, log });
1213
+ } catch (error) {
1214
+ // Valkey did not answer. reserveDollars tried to give back what it had added, key by key (a key it could
1215
+ // not is logged, dollar_giveback_error), so nothing is held by this job; the job-count
1216
+ // slots are refunded by the never-started arm below, and the job is retried: nothing started.
1217
+ log("dollar_reserve_error", { code: typeof error?.code === "string" ? error.code : "error" });
1218
+ throw new InfraRetry("the dollar windows could not be reserved", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
1219
+ }
1220
+ if (!reservation.allowed) {
1221
+ // Refused, and every dollar key it touched given back, best effort per key (the departure from the job-count
1222
+ // rule, see dollar-budget.mjs; a key that could not be is logged and the record says floor). Every held
1223
+ // job-count slot goes back too, last first: a ledger that did not issue the refusal gives back, the rule
1224
+ // the count ledgers follow among themselves above.
1225
+ // WHOSE NUMBER bound (issue #504 part B): `allocation-cap` when the refusing window's cap came from the applied split
1226
+ // or the envelope total (`capSource`), `dollar-cap` when it was the operator's own (a tie is the operator's). One
1227
+ // ledger and one window refused, and the reservation names both.
1228
+ const capSourceOf = (prefix) => (prefix === DOLLAR_KEY_PREFIX ? dollarCapSource : [scopedDollars, projectDollars, otherDollars].find((l) => l?.keyPrefix === prefix)?.capSource);
1229
+ const byAllocation = capSourceOf(reservation.ledger)?.[reservation.window] === "allocation";
1230
+ const refusalReason = byAllocation ? ALLOCATION_CAP_REASON : DOLLAR_CAP_REASON;
1231
+ const refunded = await refundLedgers(refusalReason);
1232
+ // The window is named and the amounts are not: the comment's reader may be an issue author, and the
1233
+ // amounts are the operator's, which the log carries. Which ledger refused is named too: the deployment,
1234
+ // the repo (a forge scope IS the repo the comment posts on), "this folder" (a local scope is a host path,
1235
+ // kept out of the comment and the log), or the model (operator configuration, never payload). The log
1236
+ // names a scoped or model ledger by its key prefix, a hash, never the scope string.
1237
+ const period = { day: "today's", week: "this week's", month: "this month's" }[reservation.window] ?? "a";
1238
+ // Two different states, told apart (PR #549's review): a window whose cap is BELOW one job's cap refuses every
1239
+ // such job until the operator changes a setting, which "no room left" would hide behind a wait that never ends.
1240
+ const capBelowJob = reservation.capMicros < job.maxCostMicros;
1241
+ // A project window (issue #499 part B) is "this project", for the job-count refusal's reason.
1242
+ const refusedBy = reservation.ledger === DOLLAR_KEY_PREFIX ? "deployment" : modelPrefixes.has(reservation.ledger) ? "model" : reservation.ledger === projectDollars?.keyPrefix ? "project" : reservation.ledger === otherDollars?.keyPrefix ? "other" : "scope";
1243
+ const whose = refusedBy === "deployment" ? "this deployment" : refusedBy === "model" ? `the model ${modelPrefixes.get(reservation.ledger)}` : refusedBy === "project" ? "this project" : refusedBy === "other" ? "the work outside the budget split's projects" : job.kind === "local" ? "this folder" : unqualifiedScope(scopedDollars.scope);
1244
+ // The allocation's own words (issue #504 part B): the number that bound is the split inside the operator's envelope,
1245
+ // which moves with the next priorities plan or an envelope edit, so this text never tells its reader to raise a budget.
1246
+ const adjective = { day: "daily", week: "weekly", month: "monthly" }[reservation.window] ?? "";
1247
+ const allocationText = capBelowJob
1248
+ ? `Refused: the ${adjective} share of the budget split for ${whose} is smaller than this run's cost limit, so no container was started and nothing was spent. The split changes with the next priorities plan or an envelope change. Not run.`
1249
+ : `Refused: ${period} share of the budget split for ${whose} has no room left for this run's cost limit, so no container was started and nothing was spent. Not run.`;
1250
+ await comment(
1251
+ job,
1252
+ byAllocation
1253
+ ? allocationText
1254
+ : capBelowJob
1255
+ ? `Refused: the ${{ day: "daily", week: "weekly", month: "monthly" }[reservation.window] ?? ""} dollar budget for ${whose} is smaller than this run's cost limit, so no run with this limit can start until the operator raises the budget or lowers the limit. No container was started and nothing was spent. Not run.`
1256
+ : `Refused: ${period} dollar budget for ${whose} has no room left for this run's cost limit, so no container was started and nothing was spent. Not run.`,
1257
+ );
1258
+ log("over_dollar_budget", { ledger: refusedBy, ...(refusedBy === "deployment" ? {} : { key: reservation.ledger }), ...(refusedBy === "model" ? { model: modelPrefixes.get(reservation.ledger) } : {}), window: reservation.window, reservedMicros: reservation.reservedMicros, capMicros: reservation.capMicros, amountMicros: job.maxCostMicros, ...(capBelowJob ? { capBelowJob: true } : {}), ...(byAllocation ? { source: "allocation" } : {}), refunded });
1259
+ const stranded = reservation.stranded > 0;
1260
+ const modelBasis = modelPrefixes.size === 0 ? null : stranded ? "floor" : "refunded";
1261
+ return { outcome: "policy", reason: refusalReason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: globalHeld(), dollars: stranded ? dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: job.maxCostMicros, basis: "floor", modelBasis }) : dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: 0, basis: "refunded", modelBasis }) }; // return => not retried
1262
+ }
1263
+ dollarHold = reservation.hold;
1264
+ dollars = dollarsRecord({ reservedMicros: job.maxCostMicros, settledMicros: job.maxCostMicros, basis: "floor", modelBasis: modelPrefixes.size > 0 ? "floor" : null });
919
1265
  }
920
- // budgetReserved true: the slot is reserved above and kept (a refused reservation still counts). Both
921
- // over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
922
- return { outcome: "policy", reason: budget.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
923
1266
  }
924
1267
 
925
1268
  // The user the gate above decided is the user that runs: one answer, never two call sites that agree.
926
- const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason } = await runContainer({ job, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true });
1269
+ // Issue #545: an image that declares `exitAuth` is handed a per-job key on stdin and signs its exit line with it, and
1270
+ // then only a signed line is read (run-container.mjs, run-history.mjs `authenticExitLines`). An image that does not
1271
+ // declare it is read as before, under the #542 trust rule below alone.
1272
+ const exitAuth = (img.capabilities ?? []).includes(EXIT_AUTH_CAPABILITY);
1273
+ const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason, exitWhy = null, exitLineCode = null, exitAuth: exitAuthResult = null } = await runContainer({ job: containerJob, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true, ...(modelEndpoints?.endpoints?.length > 0 ? { modelEndpoints } : {}), ...(exitAuth ? { exitAuth: true } : {}) });
927
1274
  containerRan = true;
928
- log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}) });
1275
+ // `exitAuth: "unverified"` is a run whose image signs its exit line and no signed line was found: the runner died
1276
+ // before writing one, or a line was forged or taken off the pipe. Its tokens read as unknown and its dollars settle
1277
+ // at the floor, the same as a container that wrote no exit line at all.
1278
+ log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}), ...(exitAuthResult !== null ? { exitAuth: exitAuthResult } : {}) });
929
1279
 
930
1280
  // Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
931
1281
  // so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
@@ -938,12 +1288,53 @@ export async function runJob(job, deps) {
938
1288
  await recordSpend(redis, tokensSpent, { now }).catch((err) => log("token_spend_error", { reason: err?.message }));
939
1289
  }
940
1290
 
1291
+ // SETTLE the dollar reservation (issue #501, part 4), ONCE, here, for every exit where the container ran:
1292
+ // completed, runner policy, a worker abort or cancel, exit 1, an unknown exit, and a detached container. A
1293
+ // never-started exit is not settled: the catch refunds it whole, and `isNeverStartedExit` is the one answer both
1294
+ // ask. Before any classification below, so every return and every throw carries the same `dollars`.
1295
+ const neverStarted = isNeverStartedExit({ code, aborted, detached }, neverStartedExits(job));
1296
+ if (dollarHold !== null && !neverStarted) {
1297
+ // The exit line is trusted only when the container exited ON ITS OWN (the worker did not abort, cancel, time
1298
+ // out or detach it, so the runner had the chance to write its genuine last line) AND that line's own `code` is
1299
+ // the container's real exit code (PR #542's review, round 3: a job's tool can forge a $0 line before a stop).
1300
+ const trusted = !aborted && detached !== true && Number.isSafeInteger(exitLineCode) && exitLineCode === code;
1301
+ const reservedMicros = dollarHold.amountMicros;
1302
+ const { settledMicros, basis } = dollarSettlement({ tokens, usage: usage ?? null, reservedMicros, trusted });
1303
+ // The deployment, the repo or folder and the project windows settle to the job's cost; each model window to its
1304
+ // own row of the usage ledger (`modelDollarSettlement`), never to the job's total.
1305
+ const jobPart = holdPart(dollarHold, (prefix) => jobCostPrefixes.has(prefix));
1306
+ const { applied, of } = await settleDollars(redis, jobPart, settledMicros, { log });
1307
+ let modelBasis = null;
1308
+ for (const part of dollarHold.ledgers ?? []) {
1309
+ const ref = modelPrefixes.get(part.keyPrefix);
1310
+ if (ref === undefined) continue;
1311
+ const m = modelDollarSettlement({ ref, basis, tokens, usage: usage ?? null, reservedMicros, trusted });
1312
+ const done = await settleDollars(redis, { amountMicros: reservedMicros, keys: part.keys }, m.settledMicros, { log });
1313
+ // The deployment rule below, per model window: a fault that adjusted none of its keys left the reservation.
1314
+ const mBasis = done.applied === 0 && done.of > 0 && m.settledMicros !== reservedMicros ? "floor" : m.basis;
1315
+ modelBasis = modelBasis === "floor" || mBasis === "floor" ? "floor" : "metered";
1316
+ // The model is operator configuration (a scoped-limits row), and the key a hash; no value but the amounts.
1317
+ log("dollar_model_settled", { model: ref, key: part.keyPrefix, settledMicros: m.settledMicros, basis: mBasis });
1318
+ }
1319
+ // A fault that adjusted no key leaves the whole reservation in every window, which is a floor whatever the
1320
+ // basis would have been, and the record says so. A partial one is logged (dollar_settle_error) and recorded
1321
+ // as computed: the keys that were adjusted hold the settled amount.
1322
+ dollars = applied === 0 && of > 0 && settledMicros !== reservedMicros ? dollarsRecord({ reservedMicros, settledMicros: reservedMicros, basis: "floor", modelBasis }) : dollarsRecord({ reservedMicros, settledMicros, basis, modelBasis });
1323
+ dollarHold = null;
1324
+ log("dollar_settled", { reservedMicros: dollars.reservedMicros, settledMicros: dollars.settledMicros, basis: dollars.basis, ...(modelBasis !== null ? { modelBasis } : {}) });
1325
+ } else if (dollars?.basis === "unreserved") {
1326
+ // Nothing was held, so nothing is written. A metered cost above 0 here would mean the runner's cap of 0 let a
1327
+ // priced call through, which it cannot by construction; it is logged so it cannot pass unseen.
1328
+ const spent = meteredMicros(tokens?.cost);
1329
+ if (spent !== null && spent > 0) log("dollar_unreserved_spent", { meteredMicros: spent });
1330
+ }
1331
+
941
1332
  // Issue #345: `docker run` exited with a never-started code, but THIS attempt's container was found by its cidfile,
942
1333
  // still there, and was stopped and removed (run-container.mjs, measured on Podman with its API service killed
943
1334
  // mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
944
1335
  // BEFORE the exit-code switch, where the same code would read as a free never-started exit.
945
1336
  if (detached === true) {
946
- throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1337
+ throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
947
1338
  }
948
1339
 
949
1340
  // A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
@@ -961,7 +1352,7 @@ export async function runJob(job, deps) {
961
1352
  // Awaited bare like every determinate refusal above: the adapter never throws by contract, and
962
1353
  // the one swallowed comment in this file (the catch's) justifies itself by its position.
963
1354
  await comment(job, TERMINAL_COMMENTS[reason]);
964
- return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
1355
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}) };
965
1356
  }
966
1357
 
967
1358
  switch (code) {
@@ -972,6 +1363,10 @@ export async function runJob(job, deps) {
972
1363
  // over-budget, infra): an InfraRetry job is retried, so chaining there would double-enqueue.
973
1364
  // collectChain never throws; chainEnqueued/chainRefused are additive telemetry only.
974
1365
  const chain = await collectChain({ job, prepared });
1366
+ // Issue #505: the plan, after the chain and on this branch only, for the reason the chain is here: a policy or
1367
+ // infra exit collects nothing, and a retried job must not apply what its failed attempt wrote. `portfolio` is
1368
+ // the pickup's decision; the collector asks the live file again, and both must agree. Never throws.
1369
+ const plan = await collectPlan({ job, prepared, portfolio });
975
1370
  // COMPLETED-ONLY PROMOTION, and the exclusivity is the point rather than an optimisation.
976
1371
  // A policy or infra exit leaves the canonical transcript byte-identical to what it was
977
1372
  // before this run, so a retry starts from exactly what the first attempt did -- promote on
@@ -995,6 +1390,10 @@ export async function runJob(job, deps) {
995
1390
  budgetReserved: true,
996
1391
  chainEnqueued: chain.enqueued,
997
1392
  chainRefused: chain.refused,
1393
+ ...(dollars ? { dollars } : {}),
1394
+ // Only when the collector returned a plan (a file, or a confirmed portfolio job with none, `plan-absent`): the
1395
+ // record's `plan` is null otherwise, and every other result is unchanged.
1396
+ ...(plan ? { plan } : {}),
998
1397
  };
999
1398
  }
1000
1399
  case EXIT_POLICY: {
@@ -1017,7 +1416,11 @@ export async function runJob(job, deps) {
1017
1416
  // Every other reason the runner gives still reads as runner-policy.
1018
1417
  const reason = RUNNER_POLICY_REASONS.has(exitReason) && code === 2 ? exitReason : "runner-policy";
1019
1418
  await comment(job, TERMINAL_COMMENTS[reason]);
1020
- return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
1419
+ // Issue #507: which rule of the cost guard refused (`unboundable`, `external`, `over-cap`), as the record's
1420
+ // `why`. parseExitWhy keeps only a member of the closed COST_CAP_WHYS off a `cost-cap` line that said code 2,
1421
+ // and it rides only beside a `cost-cap` reason, so a forged line can at worst name the wrong rule of three.
1422
+ const why = reason === "cost-cap" && COST_CAP_WHYS.includes(exitWhy) ? { why: exitWhy } : {};
1423
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true, ...(dollars ? { dollars } : {}), ...why };
1021
1424
  }
1022
1425
  case EXIT_INFRA:
1023
1426
  // NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
@@ -1025,7 +1428,7 @@ export async function runJob(job, deps) {
1025
1428
  // the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
1026
1429
  // BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
1027
1430
  // function never sees (issue #288).
1028
- throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1431
+ throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1029
1432
  default:
1030
1433
  // THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
1031
1434
  // (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
@@ -1039,10 +1442,11 @@ export async function runJob(job, deps) {
1039
1442
  // silently wrong for any venue where 125 is a real runner exit, and the assumption was
1040
1443
  // invisible while there was one runtime. An adapter declares its own set, or declares none
1041
1444
  // and normalises to this outcome itself.
1042
- if ((neverStartedExits(job) ?? []).includes(code)) {
1445
+ if (neverStarted) {
1446
+ // No `dollars` here: the hold is still standing, and the catch refunds it whole with the job-count slots.
1043
1447
  throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1044
1448
  }
1045
- throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
1449
+ throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), dollars });
1046
1450
  }
1047
1451
  } catch (e) {
1048
1452
  // A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
@@ -1059,35 +1463,50 @@ export async function runJob(job, deps) {
1059
1463
  // this project exists to make legible.
1060
1464
  //
1061
1465
  // Refunding is right BECAUSE no container started: identical to `container-never-started`, and the
1062
- // same both-or-neither pair, under the same `reserved`/`scopedReserved` guards so it still cannot
1063
- // double-release. The gate above catches the provider case for free, before the mint and the clone;
1466
+ // same all-or-none refund of every held job-count ledger, through the same `releaseLedgers` list, so it still
1467
+ // cannot double-release. The gate above catches the provider case for free, before the mint and the clone;
1064
1468
  // this is the backstop for every other config throw that can still land here (an unknown forge kind in
1065
1469
  // `buildContainerEnv`, a prepare-time refusal), which would otherwise keep the same slot silently.
1066
1470
  // `!containerRan` is the discriminator, and it is what makes the refund and the sentence below TRUE
1067
1471
  // rather than merely true today. Every config-tagged throw site in the worker is pre-container, so this
1068
1472
  // changes nothing now; the day one is added after a paid run, that run keeps its slot and falls through
1069
1473
  // to the untagged path instead of being refunded and publicly declared free.
1474
+ // Issue #501: a dollar hold still standing here was neither settled nor refunded. Two kinds of throw give it back
1475
+ // whole, because no container ran: never-started (the arm below) and config-refused (the next arm). Any other
1476
+ // throw may have followed a container that ran (runContainer itself throwing, a defect), so the hold STAYS, the
1477
+ // floor, which errs toward overcounting, and `dollar_hold_unsettled` says so. Released FIRST and never throwing
1478
+ // (releaseDollars), so a Valkey fault cannot replace either arm's classification.
1479
+ if (dollarHold !== null) {
1480
+ const hold = dollarHold;
1481
+ dollarHold = null;
1482
+ if ((e?.piDispatchConfig === true && !containerRan) || isNeverStartedRetry(e)) {
1483
+ const { applied, of } = await releaseDollars(redis, hold, { log });
1484
+ const heldModel = (hold.ledgers ?? []).some((l) => modelPrefixesOf(modelDollars).has(l.keyPrefix));
1485
+ dollars = dollarsRecord({ reservedMicros: hold.amountMicros, settledMicros: applied === of ? 0 : hold.amountMicros, basis: applied === of ? "refunded" : "floor", modelBasis: heldModel ? (applied === of ? "refunded" : "floor") : null });
1486
+ } else {
1487
+ log("dollar_hold_unsettled", { reservedMicros: hold.amountMicros, error: e instanceof InfraRetry ? "infra-retry" : (e?.name ?? "error") });
1488
+ const heldModel = (hold.ledgers ?? []).some((l) => modelPrefixesOf(modelDollars).has(l.keyPrefix));
1489
+ dollars = dollarsRecord({ reservedMicros: hold.amountMicros, settledMicros: hold.amountMicros, basis: "floor", modelBasis: heldModel ? "floor" : null });
1490
+ }
1491
+ }
1492
+ // The record reads `dollars` off the error on every throw path (buildRecord), so a retried or failed attempt says
1493
+ // what its windows were charged.
1494
+ if (dollars !== null && e !== null && typeof e === "object") {
1495
+ try {
1496
+ e.dollars = dollars;
1497
+ } catch {
1498
+ // a frozen error keeps its own fields; the log lines above are the record of the hold
1499
+ }
1500
+ }
1501
+
1070
1502
  if (e?.piDispatchConfig === true && !containerRan) {
1071
- // GUARDED, and the flags are the record of what actually happened. `releaseBudget` is a loop of
1503
+ // GUARDED, and the `held` list is the record of what actually happened. `releaseLedgers` is a loop of
1072
1504
  // DECRs over the active windows and can reject part-way (a read-only replica, a dropped
1073
1505
  // connection), which would otherwise replace this determinate refusal with a Redis message: the
1074
1506
  // operator would be told their queue is broken when their deployment is misconfigured, and the
1075
1507
  // escaping error is untagged so it is not retried either. A refund that did not land must not be
1076
1508
  // reported as one, so `budgetReserved` follows the ledger and not the intent.
1077
- let refunded = true;
1078
- try {
1079
- if (reserved) {
1080
- await releaseBudget(redis, { caps, now });
1081
- reserved = false;
1082
- }
1083
- if (scopedReserved && scopedCaps) {
1084
- await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
1085
- scopedReserved = false;
1086
- }
1087
- } catch (releaseError) {
1088
- refunded = false;
1089
- log("budget_release_failed", { at: "config-refused", code: releaseError?.code ?? null });
1090
- }
1509
+ const refunded = await refundLedgers("config-refused");
1091
1510
  // A FIXED sentence, and NO message in the log either. Two different reasons, both load-bearing:
1092
1511
  // `credentialFromPiAuth` puts `auth.json`'s location in its refusals and `prepare-local` puts the
1093
1512
  // operator's folder in its own, which `buildRecord` reduces to a basename precisely because a host
@@ -1103,26 +1522,24 @@ export async function runJob(job, deps) {
1103
1522
  // shipped adapter never throws; this makes that a property of the arm rather than of the wiring.
1104
1523
  await comment(job, "Refused: this deployment is misconfigured, so the job could not be started. Ask the operator to run `pi-dispatch doctor`. Not run.").catch(() => {});
1105
1524
  log("refused_config", { kind: job.kind ?? null, refunded });
1106
- // budgetReserved reflects the LEDGER: false when the refund landed, true when it did not and the
1107
- // slot is still out there.
1108
- return { outcome: "policy", reason: "config-refused", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: !refunded }; // return => not retried
1525
+ // budgetReserved reflects the LEDGER: whether the GLOBAL slot is still held after the refund (`globalHeld`).
1526
+ return { outcome: "policy", reason: "config-refused", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: globalHeld(), ...(dollars ? { dollars } : {}) }; // return => not retried
1109
1527
  }
1110
1528
 
1111
1529
  // A spawn fault (docker daemon down / binary missing) reserved a slot but never started a
1112
1530
  // container, so nothing was spent -- give the slot back before the retry. Every other throw
1113
1531
  // here (exit-1 infra, unknown exit) means the container ran and legitimately spent its slot,
1114
- // so `reason` gates the release to the never-started case only. Guarded on `reserved` and run
1115
- // once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
1116
- // budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
1117
- // true for a real container that ran and spent (exit-1 infra / unknown exit).
1118
- if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
1119
- if (e instanceof InfraRetry && e.reason === "container-never-started") {
1120
- // Both-or-neither (issue #242): a never-started container can only follow BOTH reserves (the
1121
- // scoped one precedes the global one, and the container follows both), so they refund
1122
- // together -- and a scoped refusal returned above without ever touching the global ledger.
1123
- if (reserved) await releaseBudget(redis, { caps, now });
1124
- if (scopedReserved && scopedCaps) await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
1532
+ // so `reason` gates the release to the never-started case only. It releases only what `held` still
1533
+ // lists and run once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
1534
+ if (isNeverStartedRetry(e)) {
1535
+ // All-or-none (issues #242 and #499 part B): a never-started container follows EVERY job-count reserve, so
1536
+ // every held ledger refunds together, last first, through the one helper -- and a scoped or project refusal
1537
+ // returned above with the ledgers before it already given back.
1538
+ await refundLedgers("container-never-started");
1125
1539
  }
1540
+ // After the refund, so it says what the ledger holds: false when never-started gave the global slot back, true
1541
+ // for a real container that ran and spent (exit-1 infra / unknown exit) or a refund that did not land.
1542
+ if (e instanceof InfraRetry) e.budgetReserved = globalHeld();
1126
1543
  throw e;
1127
1544
  } finally {
1128
1545
  if (prepared) await cleanup(prepared).catch(() => {});
@@ -1198,8 +1615,17 @@ const SECRET_FAILURES = {
1198
1615
  nul: "printed a value containing a NUL byte, which cannot survive the container's argv",
1199
1616
  };
1200
1617
 
1618
+ /**
1619
+ * Is this throw the "the runner never ran" retry (issue #227), whichever path raised it: a never-started exit
1620
+ * (`isNeverStartedExit`), a spawn fault, or a pre-start gate? The catch's refunds (the job-count slots and, issue
1621
+ * #501, the dollar hold) key off this one test.
1622
+ */
1623
+ function isNeverStartedRetry(e) {
1624
+ return e instanceof InfraRetry && e.reason === "container-never-started";
1625
+ }
1626
+
1201
1627
  export class InfraRetry extends Error {
1202
- constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved } = {}) {
1628
+ constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved, dollars } = {}) {
1203
1629
  super(message, cause ? { cause } : undefined);
1204
1630
  this.name = "InfraRetry";
1205
1631
  this.piDispatchRetry = true;
@@ -1218,6 +1644,9 @@ export class InfraRetry extends Error {
1218
1644
  this.provider = provider ?? null;
1219
1645
  this.model = model ?? null;
1220
1646
  this.budgetReserved = budgetReserved ?? null;
1647
+ // Issue #501: the dollar reservation's outcome (`dollarsRecord`), or null when no dollar window applied. Set by
1648
+ // the processor on a throw after the reservation, so a retried attempt's record says what its window was charged.
1649
+ this.dollars = dollars ?? null;
1221
1650
  }
1222
1651
  }
1223
1652