@edgehero/pi-dispatch 2.1.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/.env.example +41 -5
  2. package/README.md +11 -5
  3. package/deploy/docker-compose.yml +12 -0
  4. package/deploy/egress-proxy.conf +28 -3
  5. package/deploy/pi-dispatch-egress-proxy.container +8 -2
  6. package/package.json +8 -1
  7. package/src/allocation.mjs +731 -0
  8. package/src/backends.mjs +243 -0
  9. package/src/budget.mjs +40 -4
  10. package/src/cli.mjs +222 -11
  11. package/src/config.mjs +126 -5
  12. package/src/daemon-facts.mjs +3 -0
  13. package/src/deployment-venue.mjs +1 -0
  14. package/src/doctor.mjs +2261 -203
  15. package/src/dollar-budget.mjs +373 -0
  16. package/src/dollar-fingerprint.mjs +83 -0
  17. package/src/egress-cli.mjs +316 -0
  18. package/src/egress-proxy-state.mjs +35 -5
  19. package/src/egress.mjs +12 -0
  20. package/src/env-allowlist.mjs +107 -6
  21. package/src/env-file.mjs +194 -25
  22. package/src/envelope.mjs +413 -0
  23. package/src/exit-code.mjs +22 -0
  24. package/src/fleet-lease.mjs +85 -25
  25. package/src/get-token.mjs +16 -5
  26. package/src/git-dirty.mjs +67 -0
  27. package/src/github-app-setup.mjs +6 -3
  28. package/src/github-host.mjs +5 -3
  29. package/src/identity.mjs +2 -1
  30. package/src/image-preflight.mjs +98 -24
  31. package/src/image-ref.mjs +37 -0
  32. package/src/import-pi.mjs +4 -2
  33. package/src/index.mjs +407 -62
  34. package/src/init.mjs +18 -0
  35. package/src/job-id.mjs +26 -3
  36. package/src/live-probes.mjs +24 -9
  37. package/src/model-catalog.mjs +297 -0
  38. package/src/model-endpoints.mjs +649 -0
  39. package/src/model-ref.mjs +151 -0
  40. package/src/models-json.mjs +262 -0
  41. package/src/money.mjs +144 -0
  42. package/src/octokit-log.mjs +65 -0
  43. package/src/outbox-plan.mjs +218 -0
  44. package/src/outbox.mjs +29 -9
  45. package/src/output-cap.mjs +157 -0
  46. package/src/pause-windows.mjs +81 -2
  47. package/src/pi-model-loader.mjs +77 -0
  48. package/src/podman-stack.mjs +16 -3
  49. package/src/portfolio-snapshot.mjs +304 -0
  50. package/src/prepare-local.mjs +247 -12
  51. package/src/prepare.mjs +35 -3
  52. package/src/priorities.mjs +569 -0
  53. package/src/processor.mjs +599 -170
  54. package/src/project-id.mjs +17 -0
  55. package/src/projects.mjs +238 -0
  56. package/src/provider-steering.mjs +179 -65
  57. package/src/queue.mjs +111 -6
  58. package/src/reserved-env.mjs +30 -0
  59. package/src/run-container.mjs +59 -5
  60. package/src/run-history.mjs +379 -24
  61. package/src/run-mirror.mjs +30 -0
  62. package/src/runtime-settings.mjs +104 -9
  63. package/src/schedules.mjs +33 -1
  64. package/src/scoped-limits.mjs +447 -27
  65. package/src/service.mjs +15 -4
  66. package/src/session-store.mjs +131 -6
  67. package/src/start.mjs +528 -40
  68. package/src/triggers-file.mjs +65 -4
  69. package/src/triggers.mjs +135 -7
  70. package/src/up.mjs +308 -34
  71. package/src/valkey-endpoint.mjs +3 -2
package/src/start.mjs CHANGED
@@ -11,6 +11,7 @@ import { makeGitHubAuth } from "./get-token.mjs";
11
11
  import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING } from "./processor.mjs";
12
12
  import { transientError } from "./transient.mjs";
13
13
  import { makeGitHubHost } from "./github-host.mjs";
14
+ import { githubFailureFields } from "./octokit-log.mjs";
14
15
  import { makeGitLabAuth } from "./gitlab-auth.mjs";
15
16
  import { makeGitLabHost } from "./gitlab-host.mjs";
16
17
  import { makeForgejoAuth } from "./forgejo-auth.mjs";
@@ -18,14 +19,18 @@ import { makeForgejoHost } from "./forgejo-host.mjs";
18
19
  import { makeAzureAuth } from "./azure-auth.mjs";
19
20
  import { makeAzureHost } from "./azure-host.mjs";
20
21
  import { makeEgressPreflight } from "./egress.mjs";
21
- import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
22
+ import { checkSlotKey, endpointSlotKey, hash16, makeClaimSweeper, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
23
+ import { MAX_SLOTS, loadModelEndpoints, modelEndpointsPath, readOverlayModels } from "./model-endpoints.mjs";
24
+ import { builtinModel, checkModelsKnown } from "./model-catalog.mjs";
22
25
  import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
23
26
  import { cronFingerprint } from "./fingerprint.mjs";
24
27
  import { makeHostRegistry } from "./host-registry.mjs";
25
28
  import { makeImagePreflight } from "./image-preflight.mjs";
26
- import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
29
+ import { createWorker, JOB_TIMEOUT_MS, STALLED_FAILED_REASON } from "./index.mjs";
27
30
  import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, jobUserRefusal, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
28
31
  import { makeCollectChain } from "./outbox.mjs";
32
+ import { makeCollectPlan } from "./outbox-plan.mjs";
33
+ import { makePortfolioSnapshot } from "./portfolio-snapshot.mjs";
29
34
  import { containerPackagePaths, readStageManifest } from "./packages.mjs";
30
35
  import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
31
36
  import { listRunningSandboxes, makeSandboxNetworkSweeper, makeSandboxRuntimeWatch } from "./sandbox.mjs";
@@ -33,10 +38,14 @@ import { makeRetentionSweep } from "./retention-sweep.mjs";
33
38
  import { makeSandboxReaper } from "./sandbox-store.mjs";
34
39
  import { makeSessionStore } from "./session-store.mjs";
35
40
  import { scrubCredentials } from "./redact.mjs";
36
- import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
41
+ import { makeCheckOnceSpent, makeCheckPortfolioFlag, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
37
42
  import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "./watch-closer.mjs";
38
43
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
39
- import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
44
+ import { checkProjectRows, danglingProjectRows, dollarRowsWithoutCap, loadScopedLimits, scopeClaimRows } from "./scoped-limits.mjs";
45
+ import { escapeControls, loadProjects, projectOf, projectsFingerprint } from "./projects.mjs";
46
+ import { envelopeDigest, envelopeInsideJobPaths, loadEnvelopeChecked } from "./envelope.mjs";
47
+ import { NO_ENVELOPE_FINGERPRINT, makeAllocationAudit, makeAllocationLogReaper, makeAllocationState } from "./allocation.mjs";
48
+ import { optionalUsdMicros } from "./money.mjs";
40
49
  import { makeOnFailure } from "./on-failure.mjs";
41
50
  import { makeWaitChecker } from "./wait-check.mjs";
42
51
  import { makeWaitState } from "./wait-state.mjs";
@@ -44,17 +53,18 @@ import { hostQueueName, makeQueue } from "./queue.mjs";
44
53
  import { endpointShown, makeDockerEndpointResolver, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
45
54
  import { NETNS_KEEPER_MIN_AGE_MS, NETNS_KEEPER_YOUNG_MARGIN_MS, runtimeFromFacts } from "./netns-keeper.mjs";
46
55
  import { makeBackendRegistry, reapAll, resolveBackendName } from "./backend-registry.mjs";
47
- import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
56
+ import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, isPerMachineHost, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
48
57
  import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_INFO_TIMEOUT_MS, cachedPodmanInfo, decidePodmanJobUser, makePodmanBackend, makePodmanInfoReader, makePodmanReaper, observePodman, podmanConfRefusal, unavailableFor, podmanConfWidening, podmanJobUserRefusal, resolvePodmanImageUser } from "./backend-podman.mjs";
49
58
  import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootfulMemory, observeHost, observeRootfulConf, readRootfulService, rootfulConfRefusal, rootfulConfRetries, rootfulUnreadList, runtimeObservationKey } from "./runtime-observations.mjs";
50
59
 
51
60
  import { makeRunContainer } from "./run-container.mjs";
52
61
  import { resolveProviderCredential } from "./env-allowlist.mjs";
53
62
  import { makeSecretsResolver } from "./secrets.mjs";
54
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
55
- import { makeRunMirror } from "./run-mirror.mjs";
56
- import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
57
- import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
63
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
64
+ import { makeRunMirror, readMirroredRecord } from "./run-mirror.mjs";
65
+ import { readOverlay, resolveSettings } from "./runtime-settings.mjs";
66
+ import { usdFingerprint } from "./dollar-fingerprint.mjs";
67
+ import { authoredCron, envelopeJobPaths, loadSchedules, servedSchedules } from "./schedules.mjs";
58
68
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
59
69
 
60
70
 
@@ -164,7 +174,7 @@ export async function settleWithin(promise, ms, fallback) {
164
174
  * schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
165
175
  * `startWorker` registers so the watch dies with the worker that armed it (issue #295).
166
176
  */
167
- function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot) {
177
+ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot, afterReload = () => {}) {
168
178
  const path = config.triggersFile;
169
179
  const dir = dirname(path) || ".";
170
180
  const file = basename(path);
@@ -180,7 +190,9 @@ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot)
180
190
  if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
181
191
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
182
192
  clearTimeout(handles.timer);
183
- handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), WATCH_DEBOUNCE_MS);
193
+ // Issue #504 part B: a cron folder or a skills dir an edit adds is a new job path, so the envelope's place is
194
+ // judged again after every triggers reload.
195
+ handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }).then(() => afterReload(closer.reloadLog)), WATCH_DEBOUNCE_MS);
184
196
  });
185
197
  handles.watcher.unref?.();
186
198
  log("triggers_watching", { path });
@@ -192,7 +204,7 @@ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot)
192
204
  // changed.
193
205
  if (changedWhileArming(handles, readFile)) {
194
206
  log("triggers_reread_after_arming", { path });
195
- void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet });
207
+ void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }).then(() => afterReload(closer.reloadLog));
196
208
  }
197
209
  return closer;
198
210
  }
@@ -246,13 +258,74 @@ function watchPauseWindowsFile(config, ref, log, atBoot) {
246
258
  * last-good property carries its own test). A bad edit keeps `ref.current` untouched and logs
247
259
  * `scoped_limits_reload_invalid` -- the pause-windows posture, INT-SCOPED-LIMITS-FILE-CONTRACT.
248
260
  */
249
- export function reloadScopedLimits(config, ref, log) {
261
+ export function reloadScopedLimits(config, ref, log, deploymentCap = null, pair = null) {
250
262
  try {
251
- ref.current = loadScopedLimits(config);
263
+ const next = loadScopedLimits(config);
264
+ // Issue #499 part B: the new limits are checked against projects.json (`pair`, null only on a bare test wiring);
265
+ // a row naming a missing project keeps the last good limits, the boot rule held live.
266
+ const other = pair ? pairWith(config, next, pair, "limits") : null;
267
+ ref.current = next;
252
268
  log("scoped_limits_reloaded", { count: ref.current.length });
269
+ if (deploymentCap) warnDollarRowsWithoutCap(ref.current, deploymentCap(), log);
270
+ if (other) {
271
+ pair.projects.current = other;
272
+ log("projects_reloaded", { count: other.length, with: "scoped-limits" });
273
+ }
253
274
  } catch (err) {
254
275
  log("scoped_limits_reload_invalid", { reason: err?.message });
276
+ return;
277
+ }
278
+ // Issue #504 part B: the envelope's floors are judged against both files, so a committed edit re-judges it.
279
+ pair?.afterCommit?.();
280
+ }
281
+
282
+ /**
283
+ * The two files a project row joins (issue #499 part B): scoped-limits.json names `project:<id>`, projects.json defines
284
+ * the id. `{ limits, projects }` are the two live refs, and `deploymentCap` the merged per-job cap thunk the
285
+ * dollar-rows-without-cap warning reads (null on a bare wiring). A reload of either file is judged as a PAIR (`pairWith`), so the
286
+ * two live lists never disagree and a correct pair applies whatever order its files were saved in.
287
+ */
288
+ export function makeProjectPair(limits, projects, deploymentCap = null) {
289
+ return { limits, projects, deploymentCap };
290
+ }
291
+
292
+ /**
293
+ * Judge one side's new list (`next`, already loaded) against the other side, statelessly (issue #499 part B):
294
+ * 1. against the other file AS IT IS ON DISK: when that loads and the two agree, both are taken together, and the
295
+ * other side's list is returned for the caller to commit too (only when it differs from the live one). This is
296
+ * what makes a rename (`shop` to `store` in both files) or a project added with its row apply in EITHER save
297
+ * order: the second save sees the first file already on disk.
298
+ * 2. else against the other side's LIVE list: when they agree, this side alone is taken (null returned). This is the
299
+ * path when the other file is mid-edit and does not load.
300
+ * 3. else a `configError` naming the row, its index and both files, so the caller keeps its last good list.
301
+ */
302
+ function pairWith(config, next, pair, side) {
303
+ const limitsSide = side === "limits";
304
+ let disk = null;
305
+ try {
306
+ disk = limitsSide ? loadProjects(config) : loadScopedLimits(config);
307
+ } catch {
308
+ // The other file does not load right now: its own watcher says so. Judge against its live list alone.
255
309
  }
310
+ const agree = (limits, projects) => danglingProjectRows(limits, projects).length === 0;
311
+ if (disk !== null && (limitsSide ? agree(next, disk) : agree(disk, next))) {
312
+ const live = limitsSide ? pair.projects.current : pair.limits.current;
313
+ return JSON.stringify(disk) === JSON.stringify(live) ? null : disk;
314
+ }
315
+ const liveOther = limitsSide ? pair.projects.current : pair.limits.current;
316
+ if (limitsSide) checkProjectRows(next, liveOther, config.scopedLimitsFile, config.projectsFile ?? null);
317
+ else checkProjectRows(liveOther, next, config.scopedLimitsFile, config.projectsFile ?? null);
318
+ return null;
319
+ }
320
+
321
+ /**
322
+ * PR #549's review: a `scoped-limits.json` dollar row on a deployment with no per-job cap (env and overlay merged)
323
+ * refuses every job it applies to as `config-refused`, unless that job's trigger sets `run.maxCostUsd`. So it is a
324
+ * WARNING at load and at each reload, never a refusal: the rows by index and kind, never a scope string.
325
+ */
326
+ export function warnDollarRowsWithoutCap(limits, deploymentMaxCostUsd, log) {
327
+ const rows = dollarRowsWithoutCap(limits, deploymentMaxCostUsd);
328
+ if (rows.length > 0) log("scoped_limits_dollar_rows_without_cap", { rows });
256
329
  }
257
330
 
258
331
  /**
@@ -260,7 +333,7 @@ export function reloadScopedLimits(config, ref, log) {
260
333
  * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
261
334
  * unref'd and the returned closer stops the watch with the worker (issue #295).
262
335
  */
263
- function watchScopedLimitsFile(config, ref, log, atBoot) {
336
+ function watchScopedLimitsFile(config, ref, log, atBoot, deploymentCap = null, pair = null) {
264
337
  const path = config.scopedLimitsFile;
265
338
  const dir = dirname(path) || ".";
266
339
  const file = basename(path);
@@ -273,7 +346,7 @@ function watchScopedLimitsFile(config, ref, log, atBoot) {
273
346
  if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
274
347
  if (changed && changed !== file) return;
275
348
  clearTimeout(handles.timer);
276
- handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
349
+ handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog, deploymentCap, pair), WATCH_DEBOUNCE_MS);
277
350
  });
278
351
  handles.watcher.unref?.();
279
352
  log("scoped_limits_watching", { path });
@@ -282,7 +355,188 @@ function watchScopedLimitsFile(config, ref, log, atBoot) {
282
355
  }
283
356
  if (changedWhileArming(handles, readFile)) {
284
357
  log("scoped_limits_reread_after_arming", { path });
285
- reloadScopedLimits(config, ref, closer.reloadLog);
358
+ reloadScopedLimits(config, ref, closer.reloadLog, deploymentCap, pair);
359
+ }
360
+ return closer;
361
+ }
362
+
363
+ /**
364
+ * The projects reload (issue #499), exported apart from its watcher for `reloadScopedLimits`' reason: last-good is
365
+ * testable without fs.watch. A bad edit keeps `ref.current` and logs `projects_reload_invalid`; a good one swaps it,
366
+ * so the next pickup resolves against the new membership. A job already past its pickup keeps the id it was given.
367
+ * The reason is the loader's message, which never quotes a project's `name` (projects.mjs).
368
+ */
369
+ export function reloadProjects(config, ref, log, pair = null) {
370
+ try {
371
+ const next = loadProjects(config);
372
+ // Issue #499 part B: an edit that drops (or renames) a project a scoped-limits row still names is kept out, and the
373
+ // last good projects stay, unless scoped-limits.json on disk already agrees with it (`pairWith`).
374
+ const other = pair ? pairWith(config, next, pair, "projects") : null;
375
+ ref.current = next;
376
+ log("projects_reloaded", { count: ref.current.length });
377
+ if (other) {
378
+ pair.limits.current = other;
379
+ log("scoped_limits_reloaded", { count: other.length, with: "projects" });
380
+ // Every limits list that goes live is warned on, whichever file's reload committed it.
381
+ if (pair.deploymentCap) warnDollarRowsWithoutCap(other, pair.deploymentCap(), log);
382
+ }
383
+ } catch (err) {
384
+ // Escaped (PR #569's review): the parser's own refusals already are, and an fs error quoting the path is too.
385
+ log("projects_reload_invalid", { reason: escapeControls(err?.message) });
386
+ return;
387
+ }
388
+ // Issue #504 part B: as `reloadScopedLimits` does, so the envelope follows a projects edit in either save order.
389
+ pair?.afterCommit?.();
390
+ }
391
+
392
+ /**
393
+ * Watch the projects file (issue #499) as the scoped-limits watcher does: the directory, filtered to the one basename,
394
+ * debounced, the boot read as the arming baseline. The closer joins `extraClosers`, so the watch stops with the worker
395
+ * (DES-WATCHERS-CLOSE-WITH-THE-WORKER).
396
+ */
397
+ function watchProjectsFile(config, ref, log, atBoot, pair = null) {
398
+ const path = config.projectsFile;
399
+ const dir = dirname(path) || ".";
400
+ const file = basename(path);
401
+ const handles = { watcher: null, timer: null, closed: false };
402
+ const closer = makeWatchCloser(handles, log);
403
+ const readFile = () => readFileSync(path, "utf8");
404
+ readBeforeArming(handles, readFile, atBoot);
405
+ try {
406
+ handles.watcher = watch(dir, (_event, changed) => {
407
+ if (handles.closed) return;
408
+ if (changed && changed !== file) return;
409
+ clearTimeout(handles.timer);
410
+ handles.timer = setTimeout(() => reloadProjects(config, ref, closer.reloadLog, pair), WATCH_DEBOUNCE_MS);
411
+ });
412
+ handles.watcher.unref?.();
413
+ log("projects_watching", { path });
414
+ } catch (err) {
415
+ log("projects_watch_unavailable", { reason: err?.message });
416
+ }
417
+ if (changedWhileArming(handles, readFile)) {
418
+ log("projects_reread_after_arming", { path });
419
+ reloadProjects(config, ref, closer.reloadLog, pair);
420
+ }
421
+ return closer;
422
+ }
423
+
424
+ /**
425
+ * `envelopeJobPaths` as a thunk that keeps its last good answer (issue #504 part B): a triggers file caught mid-edit
426
+ * must not turn a containment check into a failure of its own, and the last parse that succeeded is what the running
427
+ * schedulers were built from. The first call has no last good answer and throws, which at boot refuses.
428
+ */
429
+ export function makeEnvelopeJobPaths(config, io = {}) {
430
+ let lastGood = null;
431
+ return () => {
432
+ try {
433
+ lastGood = envelopeJobPaths(config, io);
434
+ } catch (err) {
435
+ if (lastGood === null) throw err;
436
+ }
437
+ return lastGood;
438
+ };
439
+ }
440
+
441
+ /**
442
+ * The envelope reload (issue #504 part B), exported apart from its watcher for `reloadScopedLimits`' reason. `ref` is
443
+ * `{ current, digest }`. The file is judged against the LIVE projects and scoped limits (its floors name projects and
444
+ * sit under project rows) and the job paths of the moment, so a reload is paired with both files: their own reloads
445
+ * call this again once they commit (`pair.afterCommit`), and an envelope edit that needs a projects edit applies in
446
+ * either save order. A bad edit, or one that puts the file inside a job path, keeps the last good envelope and logs
447
+ * `envelope_reload_invalid`; the last good copy can never be one a job wrote, because every reload re-runs the
448
+ * containment check. A changed digest logs `envelope_reloaded` and calls `onChange` (the re-base, or the mismatch).
449
+ */
450
+ export function reloadEnvelope(config, ref, log, { projects, limits, maxCostMicros = () => null, jobPaths, onChange = () => {} }) {
451
+ let next;
452
+ try {
453
+ next = loadEnvelopeChecked(config, { projects: projects?.current ?? [], limits: limits?.current ?? [], maxCostMicros: maxCostMicros(), jobPaths: jobPaths() });
454
+ } catch (err) {
455
+ log("envelope_reload_invalid", { reason: escapeControls(err?.message) });
456
+ return;
457
+ }
458
+ const digest = next === null ? null : envelopeDigest(next);
459
+ const changed = digest !== ref.digest;
460
+ ref.current = next;
461
+ ref.digest = digest;
462
+ if (!changed) return;
463
+ log("envelope_reloaded", { digest });
464
+ onChange();
465
+ }
466
+
467
+ /**
468
+ * Watch the envelope file (issue #504 part B) as the projects watcher does: the directory, filtered to the one
469
+ * basename, debounced, the boot read as the arming baseline, the closer joining `extraClosers`.
470
+ */
471
+ function watchEnvelopeFile(config, ref, log, atBoot, ctx) {
472
+ const path = config.envelopeFile;
473
+ const dir = dirname(path) || ".";
474
+ const file = basename(path);
475
+ const handles = { watcher: null, timer: null, closed: false };
476
+ const closer = makeWatchCloser(handles, log);
477
+ const readFile = () => readFileSync(path, "utf8");
478
+ readBeforeArming(handles, readFile, atBoot);
479
+ try {
480
+ handles.watcher = watch(dir, (_event, changed) => {
481
+ if (handles.closed) return;
482
+ if (changed && changed !== file) return;
483
+ clearTimeout(handles.timer);
484
+ handles.timer = setTimeout(() => reloadEnvelope(config, ref, closer.reloadLog, ctx), WATCH_DEBOUNCE_MS);
485
+ });
486
+ handles.watcher.unref?.();
487
+ log("envelope_watching", { path });
488
+ } catch (err) {
489
+ log("envelope_watch_unavailable", { reason: err?.message });
490
+ }
491
+ if (changedWhileArming(handles, readFile)) {
492
+ log("envelope_reread_after_arming", { path });
493
+ reloadEnvelope(config, ref, closer.reloadLog, ctx);
494
+ }
495
+ return closer;
496
+ }
497
+
498
+ /**
499
+ * The model-endpoints reload (issue #503), exported apart from its watcher for `reloadScopedLimits`' reason: the
500
+ * last-good property is testable without fs.watch. A bad edit keeps `ref.current` and logs
501
+ * `model_endpoints_reload_invalid`; a good one swaps it, so a `slots` edit applies to the next pickup.
502
+ */
503
+ export function reloadModelEndpoints(config, ref, log) {
504
+ try {
505
+ ref.current = loadModelEndpoints(config);
506
+ log("model_endpoints_reloaded", { count: ref.current.length });
507
+ } catch (err) {
508
+ log("model_endpoints_reload_invalid", { reason: err?.message });
509
+ }
510
+ }
511
+
512
+ /**
513
+ * Watch the model-endpoints file (issue #503) as the scoped-limits watcher does. Armed only when the file is named
514
+ * or exists at boot: the default path is the deployment folder, and a worker started elsewhere must not watch an
515
+ * arbitrary directory for a file nobody declared. A default file created later is read at the next restart.
516
+ */
517
+ function watchModelEndpointsFile(config, ref, log, atBoot) {
518
+ const { path } = modelEndpointsPath(config);
519
+ const dir = dirname(path) || ".";
520
+ const file = basename(path);
521
+ const handles = { watcher: null, timer: null, closed: false };
522
+ const closer = makeWatchCloser(handles, log);
523
+ const readFile = () => readFileSync(path, "utf8");
524
+ readBeforeArming(handles, readFile, atBoot);
525
+ try {
526
+ handles.watcher = watch(dir, (_event, changed) => {
527
+ if (handles.closed) return;
528
+ if (changed && changed !== file) return;
529
+ clearTimeout(handles.timer);
530
+ handles.timer = setTimeout(() => reloadModelEndpoints(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
531
+ });
532
+ handles.watcher.unref?.();
533
+ log("model_endpoints_watching", { path });
534
+ } catch (err) {
535
+ log("model_endpoints_watch_unavailable", { reason: err?.message });
536
+ }
537
+ if (changedWhileArming(handles, readFile)) {
538
+ log("model_endpoints_reread_after_arming", { path });
539
+ reloadModelEndpoints(config, ref, closer.reloadLog);
286
540
  }
287
541
  return closer;
288
542
  }
@@ -330,6 +584,8 @@ export async function startWorker(
330
584
  makeLogSink: makeLogSinkFn = makeLogSink,
331
585
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
332
586
  makeRunMirror: makeRunMirrorFn = makeRunMirror,
587
+ // The fleet copy's by-id read, seamed so a wiring test can answer it without a live mirror.
588
+ readMirroredRecord: readMirroredRecordFn = readMirroredRecord,
333
589
  makeLogReaper: makeLogReaperFn = makeLogReaper,
334
590
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
335
591
  makeSandboxNetworkSweeper: makeSandboxNetworkSweeperFn = makeSandboxNetworkSweeper,
@@ -341,6 +597,7 @@ export async function startWorker(
341
597
  makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
342
598
  makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
343
599
  makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
600
+ makeClaimSweeper: makeClaimSweeperFn = makeClaimSweeper,
344
601
  makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
345
602
  makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
346
603
  // Which docker endpoint this host's CLI resolves (issue #278). A seam because the real one spawns the
@@ -391,6 +648,13 @@ export async function startWorker(
391
648
  // literal address every Valkey client of this worker then connects to. A seam: the real one reads /proc, probes
392
649
  // this host's addresses and resolves the name, none of which belongs in a wiring test.
393
650
  judgeValkey = defaultJudgeValkey,
651
+ // The scoped-limits watcher, injectable so a test can see what it is armed with (PR #549's review: the cap
652
+ // thunk the reload warning reads). Production passes nothing and gets the real watcher.
653
+ watchScopedLimits: watchScopedLimitsFn = watchScopedLimitsFile,
654
+ // The projects watcher (issue #499), injectable for the same reason: a test sees it armed and closed.
655
+ watchProjects: watchProjectsFn = watchProjectsFile,
656
+ // The envelope watcher (issue #504 part B), injectable for the same reason.
657
+ watchEnvelope: watchEnvelopeFn = watchEnvelopeFile,
394
658
  } = {},
395
659
  ) {
396
660
  const config = loadConfigFn(env);
@@ -463,7 +727,7 @@ export async function startWorker(
463
727
  // this race lives in is between these lines and the arming a thousand lines below -- the endpoint probe,
464
728
  // forge auth, the reaper, Valkey -- not the microseconds around the arming itself, which is what a first
465
729
  // attempt measured. `null` where a file is not configured, which reads as "nothing to compare".
466
- const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null };
730
+ const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null, projects: null, modelEndpoints: null, envelope: null };
467
731
  const recording = (into, path) => ({
468
732
  readFileSync: (file, enc) => {
469
733
  const text = readFileSync(file, enc);
@@ -482,6 +746,55 @@ export async function startWorker(
482
746
  // ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
483
747
  const scopedLimits = { current: loadScopedLimits(config, recording("scopedLimits", config.scopedLimitsFile)) };
484
748
 
749
+ // Issue #499: the projects file, same posture (INT-PROJECTS-FILE-CONTRACT): a bad file refuses boot with the operator
750
+ // present, a mutable ref for the live reload, [] when unset. The pickup gate reads it once per pickup, beside the
751
+ // limits snapshot, and the id it resolves there is the one the job's record carries.
752
+ const projects = { current: loadProjects(config, recording("projects", config.projectsFile)) };
753
+ // Issue #499 part B: a `project:<id>` row whose id is not a project refuses BOOT, naming the row and the id: it would
754
+ // read as a cap on a group that no job can belong to. The live reloads of either file hold the same rule (`pair`).
755
+ checkProjectRows(scopedLimits.current, projects.current, config.scopedLimitsFile, config.projectsFile ?? null);
756
+
757
+ // Issue #504 part B: the allocation envelope (INT-ENVELOPE-FILE-CONTRACT), same posture: a bad file refuses boot with
758
+ // the operator present, before any Valkey contact. It is judged against the projects and scoped limits just loaded
759
+ // (its floors name projects and sit under project rows), needs the merged per-job cost cap (each governed job
760
+ // reserves it against its share), and must lie outside every host path a job container can see, which is checked
761
+ // with this load and again with every reload. A mutable ref `{ current, digest }`; null when PI_ENVELOPE_FILE is
762
+ // unset, and then nothing below governs anything.
763
+ const envelopeCap = () => {
764
+ try {
765
+ const s = resolveSettings(config, readOverlay(config.settingsFile));
766
+ return optionalUsdMicros(s.invalid ? config.maxCostUsd : s.maxCostUsd, "maxCostUsd");
767
+ } catch {
768
+ return null;
769
+ }
770
+ };
771
+ const envelopeJobPathsNow = makeEnvelopeJobPaths(config);
772
+ const envelope = { current: null, digest: null };
773
+ if (config.envelopeFile !== null && config.envelopeFile !== undefined) {
774
+ envelope.current = loadEnvelopeChecked(config, { projects: projects.current, limits: scopedLimits.current, maxCostMicros: envelopeCap(), jobPaths: envelopeJobPathsNow() }, { io: recording("envelope", config.envelopeFile) });
775
+ envelope.digest = envelope.current === null ? null : envelopeDigest(envelope.current);
776
+ log("envelope_loaded", { digest: envelope.digest, window: envelope.current?.window ?? null, delegation: envelope.current?.delegation?.enabled === true });
777
+ }
778
+ // After a triggers reload (a cron folder or a skills dir is a job path): the envelope's place judged again. The live
779
+ // copy is kept either way, since every envelope reload re-runs the check and so never adopts a file a job could have
780
+ // written; this line is what tells the operator that a job can now reach the file.
781
+ const checkEnvelopePlace = (say = log) => {
782
+ if (!envelope.current) return;
783
+ try {
784
+ const inside = envelopeInsideJobPaths(config.envelopeFile, envelopeJobPathsNow());
785
+ if (inside) say("envelope_inside_job_path", { kind: inside.kind });
786
+ } catch (err) {
787
+ say("envelope_inside_job_path", { reason: escapeControls(err?.message) });
788
+ }
789
+ };
790
+
791
+ // Issue #503: the declared model endpoints, same posture (INT-MODEL-ENDPOINTS-FILE-CONTRACT): a bad file refuses
792
+ // boot with the operator present, a mutable ref for the live reload, [] when there is no file. Read per pickup
793
+ // from the ref, so a `slots` edit applies to the next job.
794
+ const modelEndpointsAt = modelEndpointsPath(config);
795
+ const modelEndpoints = { current: loadModelEndpoints(config, recording("modelEndpoints", modelEndpointsAt.path)) };
796
+ const watchModelEndpoints = modelEndpointsAt.explicit || atBoot.modelEndpoints !== null;
797
+
485
798
  // Issue #278: WHICH DOCKER DAEMON the job containers' credentials will travel to. Asked of the CLI at boot,
486
799
  // AFTER the free file validations above (a wedged CLI costs up to its bound, and must not delay them) and
487
800
  // BEFORE forge auth, the reaper, Valkey and the worker -- so a refusal here stops a process that built
@@ -643,7 +956,9 @@ export async function startWorker(
643
956
  // `doctor` reports as healthy. The boot posture is unchanged for a DETERMINATE failure, which is what
644
957
  // the local-only case is (no `gh` on PATH is `ENOENT`); a TRANSIENT one now leaves a re-resolver
645
958
  // behind instead of a permanent null.
646
- const forges = { github: { auth: null, host: makeHost() } };
959
+ // Issue #530: the GitHub clients take this worker's logger, so a client's own warning is one more JSON line here and
960
+ // its plain request lines (`GET /user - 401 ...`) are never printed (`octokitLog`).
961
+ const forges = { github: { auth: null, host: makeHost({ log }) } };
647
962
  // Per forge kind, what it would take to resolve its auth again: the closure, and the `idOf` its log
648
963
  // line needs. Present only while the last attempt failed transiently -- a determinate failure removes
649
964
  // it, because retrying a wrong credential is how a deployment pays to be told the same thing twice.
@@ -670,7 +985,7 @@ export async function startWorker(
670
985
  // modules throw untagged for a fetch rejection, a transient status and an unparseable body.
671
986
  const transient = err?.piDispatchConfig !== true;
672
987
  if (transient) authRetries.set(kind, { resolve: () => make(cfg), idOf });
673
- log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient });
988
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, ...githubFailureFields(err), transient });
674
989
  }
675
990
  };
676
991
 
@@ -711,7 +1026,7 @@ export async function startWorker(
711
1026
  authLastError.set(kind, err);
712
1027
  if (err?.piDispatchConfig === true) {
713
1028
  authRetries.delete(kind);
714
- log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient: false });
1029
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, ...githubFailureFields(err), transient: false });
715
1030
  } else {
716
1031
  authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
717
1032
  }
@@ -758,7 +1073,7 @@ export async function startWorker(
758
1073
  }
759
1074
  };
760
1075
 
761
- await attachAuth("github", makeAuth, config.github);
1076
+ await attachAuth("github", (cfg) => makeAuth(cfg, { log }), config.github);
762
1077
  // GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
763
1078
  // with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
764
1079
  // answer than an entry that exists and cannot authenticate.
@@ -844,6 +1159,10 @@ export async function startWorker(
844
1159
  } catch (err) {
845
1160
  log("log_reaper_skipped", { reason: scrubCredentials(err?.message) });
846
1161
  }
1162
+ // Issue #504 part B: the allocation audit files (`allocations/YYYY-MM.jsonl`) on the same retention, which the log
1163
+ // reaper above never reaches (it reaps the top-level `.log` and `.json` only). Never throws, so no double wrap.
1164
+ const reapAllocationLogs = makeAllocationLogReaper({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log });
1165
+ reapAllocationLogs();
847
1166
 
848
1167
  // REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
849
1168
  // kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
@@ -896,6 +1215,22 @@ export async function startWorker(
896
1215
  // ioredis printed a stack per reconnect attempt.
897
1216
  onValkeyError(redis, "shared client");
898
1217
 
1218
+ // Issue #504 part B: the applied split lives in Valkey (`alloc:plan`), shared by every host. Only with an envelope.
1219
+ // The boot reconcile seeds the neutral split when there is none, expires a plan past its life and re-bases on an
1220
+ // envelope this host changed while it was down; a fault is logged and the first pickup tries again, because the
1221
+ // pickup reconciles too and refuses nothing it cannot judge (it throws, and the job is retried).
1222
+ // Built on EVERY host, with an envelope or without: a host with none still asks, at each pickup, whether the fleet is
1223
+ // governed (an applied split exists), and refuses its jobs as envelope-mismatch when it is.
1224
+ const allocationState = makeAllocationState({ redis, host: config.workerName, audit: makeAllocationAudit({ logsDir: config.logsDir }), log });
1225
+ const reconcileAllocation = (why) => {
1226
+ if (!envelope.current) return Promise.resolve();
1227
+ return allocationState
1228
+ .reconcile({ envelope: envelope.current, digest: envelope.digest, now: new Date() })
1229
+ .then((r) => log("allocation_reconciled", { why, mismatch: r.mismatch, digest: envelope.digest, applied: r.state?.envelopeDigest ?? null }))
1230
+ .catch((err) => log("allocation_reconcile_failed", { why, reason: scrubCredentials(err?.message) }));
1231
+ };
1232
+ await reconcileAllocation("boot");
1233
+
899
1234
  // This host's own stale scope claims, gated on the reaper having having enumerated: the
900
1235
  // reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
901
1236
  // host is a claim for a container that no longer exists. Deleting it is not a second source of truth --
@@ -904,10 +1239,20 @@ export async function startWorker(
904
1239
  // mechanism, so a fault costs one TTL of a stale claim and never a boot.
905
1240
  try {
906
1241
  if (config.workerNameDeclared)
907
- await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current.map((r) => ({ concurrent: r.concurrent, hash: scopeKeyPrefix(r.scope).slice("budget:s:".length) })), log })({ reaped });
1242
+ await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopeClaimRows(scopedLimits.current), log })({ reaped });
908
1243
  } catch (err) {
909
1244
  log("scope_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
910
1245
  }
1246
+ // Issue #503: this host's stale model endpoint claims, on the same precondition and for the same reason (each is a
1247
+ // claim for a container). Every index up to the parse ceiling, not up to today's `slots`: `slots` can be lowered
1248
+ // live, and a claim on an index above the new value is still this host's to clear. A per-machine endpoint (a host
1249
+ // alias name) takes no fleet claim, so it has none to sweep. The sweep stops at its first fault.
1250
+ try {
1251
+ if (config.workerNameDeclared && modelEndpoints.current.length > 0)
1252
+ await makeClaimSweeperFn({ redis, workerName: config.workerName, keyFor: endpointSlotKey, rows: modelEndpoints.current.filter((e) => !isPerMachineHost(e.host)).map((e) => ({ hash: hash16(e.id), count: MAX_SLOTS })), event: "endpoint_claims", log })({ reaped });
1253
+ } catch (err) {
1254
+ log("endpoint_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
1255
+ }
911
1256
 
912
1257
  // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
913
1258
  // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
@@ -974,12 +1319,17 @@ export async function startWorker(
974
1319
  // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
975
1320
  // no job then issues a single extra Valkey command.
976
1321
  const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
977
- const recordRun = ({ job, result, error, startedAt, endedAt }) => {
1322
+ const recordRun = ({ job, result, error, startedAt, endedAt, project }) => {
1323
+ // The project (issue #499) was resolved at the pickup gate and rides here as `project` (an id or null), so a live
1324
+ // edit of projects.json mid-run cannot make the record disagree with what the job was counted against. A record
1325
+ // path that ends BEFORE the pickup gate (the wait gate's refusals) passes none, and resolves from the live ref
1326
+ // with the same function.
1327
+ const projectId = project !== undefined ? project : projectOf(job?.data ?? {}, projects.current);
978
1328
  // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
979
1329
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
980
1330
  // The default venue rides the same way and for the same reason (#277): it is the value the registry
981
1331
  // below is built with, so the record resolves a job's venue exactly as dispatch does.
982
- const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend });
1332
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId });
983
1333
  writeRecord(record);
984
1334
  // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
985
1335
  // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
@@ -997,25 +1347,30 @@ export async function startWorker(
997
1347
  // the disarm -- the same chosen direction, met at shutdown instead of a crash.
998
1348
  void disarmOnce({ job, endedAt });
999
1349
  };
1350
+ // The record read back, for a job the queue lost the lock of after it finished (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK,
1351
+ // CONST-RETRY-INFRA-ONLY): this host's own file first, then the fleet's copy where a mirror is armed, because the
1352
+ // host that meets the stalled job need not be the one that ran it. A single host has no mirror and needs none.
1353
+ const settledRecord = makeSettledRecord({
1354
+ readRecord: makeReadRecord({ logsDir: config.logsDir }),
1355
+ readMirrored: runMirror ? (jobId) => readMirroredRecordFn(redis, sanitizeJobId(jobId)) : null,
1356
+ });
1000
1357
 
1001
1358
  // INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
1002
- // closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the eight
1359
+ // closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the fourteen
1003
1360
  // effective settings from the overlay over env; an invalid overlay returns `{ invalid }` (logged loudly,
1004
1361
  // key-name-only per no-pii-in-logs) so the processor RETURNS a settings-overlay-invalid refusal instead
1005
1362
  // of the run.
1006
1363
  const settingsFile = config.settingsFile;
1007
1364
  const getSettings = () => {
1008
- const res = readOverlay(settingsFile, { log });
1365
+ // `resolveSettings` merges overlay over env and then checks the cross-key dollar rule on the MERGED values
1366
+ // (issue #501), so an overlay window with an env per-job cap is valid. `secretProfiles` rides alongside the
1367
+ // effective keys; that function says why.
1368
+ const res = resolveSettings(config, readOverlay(settingsFile, { log }));
1009
1369
  if (res.invalid) {
1010
1370
  log("settings_overlay_invalid", { reason: res.invalid, settingsFile });
1011
1371
  return { invalid: res.invalid };
1012
1372
  }
1013
- // REQ-TRIGGER-SECRETS rides ALONGSIDE the ten tunables rather than inside them. `effectiveSettings`
1014
- // resolves `overlay > env` over a fixed ten-key literal, and its own tests pin that key set and
1015
- // assert an empty overlay returns the config verbatim -- so an eleventh key there would break both,
1016
- // and would also claim a precedence this key deliberately does not have (a name declared in both
1017
- // sources is refused per delivery, not silently won by either).
1018
- return { ...effectiveSettings(config, res.overlay), secretProfiles: res.overlay?.secretProfiles ?? {} };
1373
+ return res;
1019
1374
  };
1020
1375
 
1021
1376
  // Resolve the Worker constructor's slot count once from the overlay: a present overlay may raise or lower
@@ -1023,6 +1378,13 @@ export async function startWorker(
1023
1378
  // the per-job path enforce the refusal (getSettings already logged the invalid reason).
1024
1379
  const bootSettings = getSettings();
1025
1380
  const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
1381
+ // PR #549's review: the merged per-job cap, for the scoped-limits dollar-row warning at boot and at each reload.
1382
+ // An invalid overlay falls back to the env value, the same fallback as the slot count above.
1383
+ const deploymentMaxCostUsd = () => {
1384
+ const s = getSettings();
1385
+ return s.invalid ? config.maxCostUsd : s.maxCostUsd;
1386
+ };
1387
+ warnDollarRowsWithoutCap(scopedLimits.current, bootSettings.invalid ? config.maxCostUsd : bootSettings.maxCostUsd, log);
1026
1388
 
1027
1389
  // INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
1028
1390
  // enqueues chained children onto the CRON queue via enqueueLocalJob -- this host's own when one is
@@ -1033,6 +1395,18 @@ export async function startWorker(
1033
1395
  // else would enqueue a job only this host can run onto a queue every host drains.
1034
1396
  const collectChain = makeCollectChain({ queue: cronQueue, config, log });
1035
1397
 
1398
+ // Issue #505. One live-file check, shared by the pickup gate, the snapshot and the plan collector, so the three ask one
1399
+ // file by one rule: the same file and rule as the one-shot checks below (`onceTriggersFile`). `pi-dispatch run
1400
+ // --trigger` reads it there too, so the command that fires a trigger and the check that confirms its flag see one file.
1401
+ const checkPortfolioFlag = makeCheckPortfolioFlag({ triggersPath: onceTriggersFile });
1402
+ const governingNow = () => (envelope.current ? { envelope: envelope.current, digest: envelope.digest } : null);
1403
+ // The plan collector: a completed portfolio job's /outbox/priorities.json, applied under the envelope by the same
1404
+ // allocation state every pickup reconciles. Never throws, like collectChain beside it.
1405
+ const collectPlan = makeCollectPlan({ allocation: allocationState, governing: governingNow, projects: () => projects.current, checkPortfolioFlag, log });
1406
+ // The snapshot a confirmed portfolio job reads as /job/portfolio.json, built at prepare. The run counts are complete
1407
+ // only with the run mirror, which a declared worker name arms (the `runMirror` rule below).
1408
+ const portfolioSnapshot = makePortfolioSnapshot({ checkPortfolioFlag, governing: governingNow, projects: () => projects.current, limits: () => scopedLimits.current, allocation: allocationState, redis, logsDir: config.logsDir, mirror: config.workerNameDeclared === true, log });
1409
+
1036
1410
  // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest at EACH job start, like
1037
1411
  // getSettings above and the pause-window ref below.
1038
1412
  //
@@ -1225,6 +1599,24 @@ export async function startWorker(
1225
1599
  // here is no opinion at all, and such a host must never be able to disagree with one that has one.
1226
1600
  fpCron: () => cronFingerprint(authoredCron(config), { tz: hostTz }) ?? "",
1227
1601
  cronCount: () => schedules.current.length,
1602
+ // Issue #501 part 6: a fingerprint of the dollar caps this host judges the SHARED dollar counters against (the four
1603
+ // settings as a job resolves them, and the scoped-limits dollar rows), so doctor can name two hosts that would
1604
+ // admit different jobs against one counter. A thunk for `fpCron`'s reason: the overlay and the scoped-limits file
1605
+ // change without a restart. Read without a log, so an invalid overlay is not logged on every beat (each job
1606
+ // logs it already). With an invalid overlay it hashes the env values, the slot count's fallback above; such a
1607
+ // host refuses every job (settings-overlay-invalid) until the file is fixed. The env list rides along: it decides
1608
+ // which model rows a job without its own list reserves in.
1609
+ fpUsd: () => {
1610
+ const settings = resolveSettings(config, readOverlay(settingsFile));
1611
+ return usdFingerprint(settings.invalid ? config : settings, scopedLimits.current, config.allowedModels);
1612
+ },
1613
+ // Issue #499 part C: a fingerprint of the LIVE projects (ids and member hashes, never a name), so doctor can name a
1614
+ // host whose projects.json differs. Each host resolves its own jobs' project from its own copy, while the project
1615
+ // rows' counters are shared, so two copies put one repo in two projects. A thunk, so a live edit shows in one beat.
1616
+ fpProjects: () => projectsFingerprint(projects.current),
1617
+ // Issue #504 part B: the digest of this host's live envelope (`envelopeDigest`, 16 hex, never a value), so doctor can
1618
+ // name a host whose envelope differs; such a host refuses governed jobs as `envelope-mismatch`. `none` without one.
1619
+ fpEnvelope: () => envelope.digest ?? NO_ENVELOPE_FINGERPRINT,
1228
1620
  });
1229
1621
 
1230
1622
 
@@ -1506,6 +1898,9 @@ export async function startWorker(
1506
1898
  // `operator-cancel`, because the operator initiated it and a push telling them what they just did is
1507
1899
  // noise with a pager attached.
1508
1900
  const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", ...RUNNER_POLICY_REASONS]);
1901
+ // One predicate for the completed listener and the lost-lock path below, so a record replays exactly the page its
1902
+ // result would have sent.
1903
+ const pagesAsPolicy = (result) => Boolean(onFailure) && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason) && result.budgetReserved !== false;
1509
1904
  // The infra-terminal sentence (issue #288). FIXED, never err.message: the message classes that reach
1510
1905
  // a failedReason carry host paths and library words (the #310 record), and for a local job this text
1511
1906
  // lands verbatim in the service log through the adapter's stdout fallthrough. The worker log already
@@ -1542,6 +1937,7 @@ export async function startWorker(
1542
1937
  getSettings,
1543
1938
  redis,
1544
1939
  recordRun,
1940
+ settledRecord,
1545
1941
  extraClosers,
1546
1942
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
1547
1943
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
@@ -1549,6 +1945,12 @@ export async function startWorker(
1549
1945
  // Issue #242: the scoped-limits snapshot the pickup gate and the scoped budget read, once per
1550
1946
  // pickup, from the live-reloaded ref -- same next-job grain as pauseUntil above.
1551
1947
  scopedLimits: () => scopedLimits.current,
1948
+ // Issue #499: the projects snapshot, read by the pickup gate once, beside the limits snapshot above.
1949
+ projects: () => projects.current,
1950
+ // Issue #504 part B: the live envelope and its digest, and the reconcile the pickup runs before it narrows a job's
1951
+ // dollar ledgers by the applied split. Without an envelope `current()` is null, and `fleetGoverned` asks whether an
1952
+ // applied split exists: if it does, this host's jobs refuse as envelope-mismatch rather than run ungoverned.
1953
+ allocation: { current: () => (envelope.current ? { envelope: envelope.current, digest: envelope.digest } : null), reconcile: allocationState.reconcile, fleetGoverned: allocationState.fleetGoverned },
1552
1954
  // Issue #230. The `after` ceiling is read per pickup from config rather than frozen into the
1553
1955
  // processor, so it is one value with one home; the wait state shares the budget's redis client
1554
1956
  // because it describes the same delayed jobs that client already reasons about.
@@ -1567,6 +1969,14 @@ export async function startWorker(
1567
1969
  // the operator limited to one. Nothing refreshes this claim, deliberately: a refresher would be a
1568
1970
  // second thing to get wrong for a window that cannot be reached.
1569
1971
  scopeLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: scopeSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
1972
+ // Issue #503: the fleet-wide half of a model endpoint's `slots`, armed like the scope lease (declaring a name is
1973
+ // declaring a fleet) and with its TTL for its reason: the claim lives as long as the job's container, which
1974
+ // `JOB_TIMEOUT_MS` bounds. Without a declared name only the in-process bound applies, per host.
1975
+ endpointLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: endpointSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
1976
+ // The endpoint gate's snapshot, read per pickup: the live-reloaded declaration and the overlay models.json, which
1977
+ // the operator edits without a restart too. Read only when an endpoint is declared (the gate's own rule).
1978
+ modelEndpoints: () => modelEndpoints.current,
1979
+ overlayModels: () => readOverlayModels(config.globalPiDir),
1570
1980
  checkLease: hostQueue
1571
1981
  ? makeFleetLease({
1572
1982
  redis,
@@ -1584,6 +1994,7 @@ export async function startWorker(
1584
1994
  maxFaults: () => config.waitMaxFaults,
1585
1995
  deps: {
1586
1996
  collectChain,
1997
+ collectPlan,
1587
1998
  // The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
1588
1999
  // only on a FOREIGN positive mark (index.mjs binds the real queue jobId so a retry of the
1589
2000
  // spending delivery is excused). In the compose topology this check is the once-enforcement
@@ -1600,24 +2011,41 @@ export async function startWorker(
1600
2011
  //
1601
2012
  // A PROBE: whatever it resolves is dropped on the floor. The credential itself is read where it always
1602
2013
  // was, inside buildContainerEnv, so no live key is ever in scope in the processor.
1603
- checkProviderCredential: (job) => {
2014
+ // Issue #503: `modelEndpoints` is the pickup's snapshot, the same one runContainer hands buildContainerEnv.
2015
+ checkProviderCredential: (job, { modelEndpoints = null } = {}) => {
1604
2016
  try {
1605
- resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv });
2017
+ resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv, modelEndpoints });
1606
2018
  return { ok: true };
1607
2019
  } catch (error) {
1608
2020
  // Only OUR determinate refusal. Anything else (a bug here, an fs fault the module does not model)
1609
2021
  // must not become a policy refusal on the operator's issue: it rethrows into runJob's catch, which
1610
2022
  // classifies it the way it always did.
2023
+ // Issue #503: a transient overlay read at this pickup is no verdict; the processor retries it as infra.
2024
+ if (error?.piDispatchTransient === true) return { ok: false, unavailable: error.code ?? "unreadable" };
1611
2025
  if (error?.piDispatchConfig !== true) throw error;
1612
2026
  return { ok: false, message: error.message };
1613
2027
  }
1614
2028
  },
2029
+ // Issue #502. The model-exists gate: pi's builtin catalog, then the overlay models.json, read at most once per
2030
+ // job and only when a model is not builtin, so the operator's edits apply without a restart. The SAME file
2031
+ // the endpoint gate reads, through the same reader, so absent, unreadable and unparseable mean one thing.
2032
+ checkModelsKnown: (refs) => checkModelsKnown(refs, { readOverlay: () => readOverlayModels(config.globalPiDir) }),
2033
+ // Issue #503 part 7: the builtin catalog's model object, for the zero-rated check that lets a job on local
2034
+ // zero-rated models reserve nothing in the dollar windows (processor.mjs, `zeroRatedVerdict`).
2035
+ builtinModel,
2036
+ // Issue #502. The deployment's allowed-model list (PI_ALLOWED_MODELS), null = unrestricted. Env only, and
2037
+ // handed to the processor here rather than through the settings overlay, which a model-callable tool writes.
2038
+ // index.mjs folds it into the effective job (`effectiveJobOf`) under the trigger's own `run.models`.
2039
+ allowedModels: config.allowedModels,
1615
2040
  // Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
1616
2041
  // asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
1617
2042
  // service below the version floor turns a wait into a paid run nothing can tell from a correct
1618
2043
  // one. In the compose topology the worker's read is the live inode while the receiver's is dead
1619
2044
  // until restart, which is exactly the deployment where the skew happens.
1620
2045
  checkWaitSkew: makeCheckWaitSkew({ triggersPath: onceTriggersFile }),
2046
+ // Issue #505. Whether the live file still flags a portfolio job's cron trigger, read when such a job is picked
2047
+ // up, from the same file and by the same rule as the two checks above (built once, beside collectPlan).
2048
+ checkPortfolioFlag,
1621
2049
  // Issue #230. Whether a job the supersede lease names is still in the queue. Without it a holder
1622
2050
  // that vanished by any route except the clean one leaves a key that refuses every later delivery
1623
2051
  // for that target until it expires -- and a refused forge delivery is gone, since no webhook
@@ -1694,6 +2122,11 @@ export async function startWorker(
1694
2122
  jobImage: config.jobImage,
1695
2123
  // #277: the venue a retained directory records, which the sandbox refuses by when it is not here.
1696
2124
  defaultBackend: config.defaultBackend,
2125
+ // Issue #504 part B: a local job's folder is resolved at prepare and mounted resolved; one named inside a run
2126
+ // root or a cron folder must still resolve inside it, and none may be the envelope's folder or above it.
2127
+ localPlacement: { jobPaths: envelopeJobPathsNow, envelopeFile: config.envelopeFile ?? null },
2128
+ // Issue #505: /job/portfolio.json for a confirmed portfolio job.
2129
+ portfolioSnapshot,
1697
2130
  preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
1698
2131
  // The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
1699
2132
  // per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
@@ -1762,7 +2195,7 @@ export async function startWorker(
1762
2195
  // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
1763
2196
  // what tells a human a run did nothing. The container's own output already streams via
1764
2197
  // runContainer's onOutput during the run.
1765
- // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
2198
+ // `reason` is a fixed enum (worker-abort | over-budget | dollar-cap | allocation-cap | envelope-mismatch | portfolio-no-envelope | portfolio-snapshot-oversize | unprotected-branch | runner-policy |
1766
2199
  // provider-auth-refused | job-image-missing), never
1767
2200
  // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
1768
2201
  // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
@@ -1775,11 +2208,16 @@ export async function startWorker(
1775
2208
  // and never in the failed listener -- a failed-only mount would miss exactly the paid terminals
1776
2209
  // the feature exists for. Folded into the existing listener body, never a second w.on: the
1777
2210
  // start-wiring harness records ONE handler per event, and two would race the log line's pin.
1778
- if (onFailure && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason)) {
2211
+ // `budgetReserved !== false` (issue #502): `model-not-allowed` is both a runner stop (paid, pages) and a
2212
+ // pre-spend refusal of a main model outside the job's list (free, comments, pages nobody), and the reason
2213
+ // alone cannot tell them apart. Every paid terminal above carries `budgetReserved: true`.
2214
+ if (pagesAsPolicy(result)) {
1779
2215
  onFailure({ jobId: job?.id, outcome: "policy", reason: result.reason });
1780
2216
  }
1781
2217
  });
1782
- for (const w of allWorkers) w.on("failed", (job, err) => {
2218
+ // The failed listener's body, for a failure the queue decided. Split out only so the lost-lock check below can
2219
+ // run it after an await; a failure of any other reason runs it synchronously, exactly as before.
2220
+ const onFailed = (job, err) => {
1783
2221
  log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) });
1784
2222
  // The one reason whose sentence is the fix itself (issue #458): logged WHOLE, beside the cut line above.
1785
2223
  if (err?.reason === NETNS_KEEPER_NOT_HOLDING || err?.reason === NETNS_KEEPER_CRASH_LOOP) log("job_failed_netns_keeper", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? "") });
@@ -1794,6 +2232,29 @@ export async function startWorker(
1794
2232
  void comment({ ...job.data, id: job.id }, (typeof err?.reason === "string" && Object.hasOwn(FAILED_COMMENT_BY_REASON, err.reason) ? FAILED_COMMENT_BY_REASON[err.reason] : null) ?? FAILED_COMMENT);
1795
2233
  onFailure?.({ jobId: job?.id, outcome: "failed", reason: typeof err?.reason === "string" ? err.reason : "infra" });
1796
2234
  }
2235
+ };
2236
+ // A job that FINISHED, then lost its lock (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK). When Valkey is unreachable
2237
+ // for longer than the lock renewal window while a processor finishes, the record is written, BullMQ refuses the
2238
+ // completion ("Missing lock"), and its stall check (`maxStalledCount: 0`) fails the job at the next pickup with
2239
+ // STALLED_FAILED_REASON. Without this check that job posted the failure comment and paged the operator for a run
2240
+ // that ended. So a terminal stall failure first asks the run record: when the record says this attempt finished
2241
+ // without failing, the job's terminal line is `job_lost_lock_after_completion` and nothing is posted. The page
2242
+ // the completed listener would have sent for that record (a paid policy stop) is sent here instead, because
2243
+ // that listener never saw the first finish. No record, an earlier attempt's, a `failed` one, or a lookup fault
2244
+ // keeps the failure path below: the comment and the page.
2245
+ for (const w of allWorkers) w.on("failed", (job, err) => {
2246
+ if (!(job?.finishedOn && err?.message === STALLED_FAILED_REASON)) return onFailed(job, err);
2247
+ void (async () => {
2248
+ // `attemptsMade` is read AFTER BullMQ's moveToFailed added one, so it equals the attempt number the
2249
+ // record carries (`buildRecord`'s `attemptsMade + 1`, written while the job was processing).
2250
+ // A record found and refused is said, with its fixed reason, so an operator reading the failure comment
2251
+ // can see why the record did not suppress it (a clock skew past the tolerance reads `older-than-job`).
2252
+ const onReject = (reason, source) => log("job_lost_lock_record_rejected", { jobId: job.id, reason, source });
2253
+ const record = await settledRecord(job.id, { attempt: job.attemptsMade, since: job.timestamp, onReject });
2254
+ if (!record) return onFailed(job, err);
2255
+ log("job_lost_lock_after_completion", { jobId: job.id, outcome: record.outcome, ...(record.reason ? { reason: record.reason } : {}) });
2256
+ if (pagesAsPolicy(record)) onFailure({ jobId: job.id, outcome: "policy", reason: record.reason });
2257
+ })().catch(() => {}); // `settledRecord` never rejects; this only keeps a throwing log line from going unhandled
1797
2258
  });
1798
2259
 
1799
2260
  // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
@@ -1844,7 +2305,7 @@ export async function startWorker(
1844
2305
  // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1845
2306
  // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
1846
2307
  if (config.triggersFile) {
1847
- extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers));
2308
+ extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers, checkEnvelopePlace));
1848
2309
  }
1849
2310
 
1850
2311
  // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
@@ -1853,9 +2314,29 @@ export async function startWorker(
1853
2314
  extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log, atBoot.pauseWindows));
1854
2315
  }
1855
2316
 
2317
+ // Issue #499 part B: the two files a project row joins reload as a pair; the deployment cap rides along so a limits
2318
+ // list either reload commits gets the dollar-rows-without-cap warning.
2319
+ const projectPair = makeProjectPair(scopedLimits, projects, deploymentMaxCostUsd);
2320
+ // Issue #504 part B: the envelope reloads with the pair, since its floors are judged against both files: a committed
2321
+ // limits or projects edit re-judges it from disk, and its own edits are judged against the live pair.
2322
+ const envelopeCtx = { projects, limits: scopedLimits, maxCostMicros: envelopeCap, jobPaths: envelopeJobPathsNow, onChange: () => void reconcileAllocation("reload") };
2323
+ if (config.envelopeFile) {
2324
+ projectPair.afterCommit = () => reloadEnvelope(config, envelope, log, envelopeCtx);
2325
+ extraClosers.push(watchEnvelopeFn(config, envelope, log, atBoot.envelope, envelopeCtx));
2326
+ }
1856
2327
  // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
1857
2328
  if (config.scopedLimitsFile) {
1858
- extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log, atBoot.scopedLimits));
2329
+ extraClosers.push(watchScopedLimitsFn(config, scopedLimits, log, atBoot.scopedLimits, deploymentMaxCostUsd, projectPair));
2330
+ }
2331
+
2332
+ // Issue #499 live edit: the projects file, keep-last-good on a bad edit.
2333
+ if (config.projectsFile) {
2334
+ extraClosers.push(watchProjectsFn(config, projects, log, atBoot.projects, projectPair));
2335
+ }
2336
+
2337
+ // Issue #503 live edit: the model endpoints, keep-last-good on a bad edit.
2338
+ if (watchModelEndpoints) {
2339
+ extraClosers.push(watchModelEndpointsFile(config, modelEndpoints, log, atBoot.modelEndpoints));
1859
2340
  }
1860
2341
 
1861
2342
  // issue #292 / OQ-007: re-run the three retention sweeps on a timer, because the supported deployment
@@ -1873,6 +2354,8 @@ export async function startWorker(
1873
2354
  const sweep = makeRetentionSweepFn({
1874
2355
  reapers: [
1875
2356
  { name: "log", reap: reapLogs },
2357
+ // Issue #504 part B: where boot runs it, right after the run history it sits beside.
2358
+ { name: "allocation_log", reap: reapAllocationLogs },
1876
2359
  { name: "sandbox", reap: reapSandboxes },
1877
2360
  { name: "session", reap: () => sessionStore.reapSessions() },
1878
2361
  ],
@@ -1912,6 +2395,11 @@ export async function startWorker(
1912
2395
  softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
1913
2396
  scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
1914
2397
  scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
2398
+ projectsFile: config.projectsFile, // issue #499: null = no projects
2399
+ projects: projects.current.length, // project count, never a name
2400
+ envelopeFile: config.envelopeFile, // issue #504: null = no envelope and no delegation
2401
+ envelopeDigest: envelope.digest, // the envelope's 16-hex digest, the host row's fpEnvelope; null without one
2402
+ modelEndpoints: modelEndpoints.current.length, // issue #503: declared model endpoints, each a slot lease at pickup
1915
2403
  image: config.jobImage,
1916
2404
  valkey: config.valkeyUrl,
1917
2405
  // Issue #464: the literal address every Valkey client of this worker dials, beside the URL as written; null