@edgehero/pi-dispatch 1.10.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +29 -2
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +363 -13
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1348 -326
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
package/src/start.mjs CHANGED
@@ -1,9 +1,15 @@
1
- import { readFileSync, watch } from "node:fs";
1
+ import { readdirSync, readFileSync, statSync, watch } from "node:fs";
2
+ import { lookup as dnsLookup } from "node:dns/promises";
3
+ import { homedir, networkInterfaces, release as osRelease, userInfo } from "node:os";
2
4
  import { dirname, basename, join } from "node:path";
3
- import { configError, loadConfig } from "./config.mjs";
4
- import { makeRedisClient, parseConnection } from "./connection.mjs";
5
+ import { configError, ensureJobsDir, ensureSandboxDir, ensureUnderAccountRoot, loadConfig } from "./config.mjs";
6
+ import { passwdNameFrom, probeTcpAddress, readSubuidRanges, resolveWorkerValkey } from "./podman-stack.mjs";
7
+ import { valkeyClientContext } from "./valkey-endpoint.mjs";
8
+ import { authRefusalFor, makeRedisClient, onValkeyError, parseConnection, valkeyAuthState, valkeyPasswordFor } from "./connection.mjs";
5
9
  import { reconcileGated, reloadSchedules } from "./cron.mjs";
6
10
  import { makeGitHubAuth } from "./get-token.mjs";
11
+ import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING } from "./processor.mjs";
12
+ import { transientError } from "./transient.mjs";
7
13
  import { makeGitHubHost } from "./github-host.mjs";
8
14
  import { makeGitLabAuth } from "./gitlab-auth.mjs";
9
15
  import { makeGitLabHost } from "./gitlab-host.mjs";
@@ -18,25 +24,34 @@ import { cronFingerprint } from "./fingerprint.mjs";
18
24
  import { makeHostRegistry } from "./host-registry.mjs";
19
25
  import { makeImagePreflight } from "./image-preflight.mjs";
20
26
  import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
27
+ import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, jobUserRefusal, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
21
28
  import { makeCollectChain } from "./outbox.mjs";
22
29
  import { containerPackagePaths, readStageManifest } from "./packages.mjs";
23
30
  import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
24
- import { listRunningSandboxes } from "./sandbox.mjs";
31
+ import { listRunningSandboxes, makeSandboxNetworkSweeper, makeSandboxRuntimeWatch } from "./sandbox.mjs";
32
+ import { makeRetentionSweep } from "./retention-sweep.mjs";
25
33
  import { makeSandboxReaper } from "./sandbox-store.mjs";
26
34
  import { makeSessionStore } from "./session-store.mjs";
35
+ import { scrubCredentials } from "./redact.mjs";
27
36
  import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
37
+ import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "./watch-closer.mjs";
28
38
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
29
39
  import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
40
+ import { makeOnFailure } from "./on-failure.mjs";
30
41
  import { makeWaitChecker } from "./wait-check.mjs";
31
42
  import { makeWaitState } from "./wait-state.mjs";
32
43
  import { hostQueueName, makeQueue } from "./queue.mjs";
33
- import { makeLocalBackend, makeReaper, makeStopContainer } from "./backend-local.mjs";
34
- import { makeBackendRegistry, reapAll } from "./backend-registry.mjs";
35
- import { DEFAULT_BACKEND } from "./backends.mjs";
44
+ import { endpointShown, makeDockerEndpointResolver, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
45
+ import { NETNS_KEEPER_MIN_AGE_MS, NETNS_KEEPER_YOUNG_MARGIN_MS, runtimeFromFacts } from "./netns-keeper.mjs";
46
+ import { makeBackendRegistry, reapAll, resolveBackendName } from "./backend-registry.mjs";
47
+ import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
48
+ import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_INFO_TIMEOUT_MS, cachedPodmanInfo, decidePodmanJobUser, makePodmanBackend, makePodmanInfoReader, makePodmanReaper, observePodman, podmanConfRefusal, unavailableFor, podmanConfWidening, podmanJobUserRefusal, resolvePodmanImageUser } from "./backend-podman.mjs";
49
+ import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootfulMemory, observeHost, observeRootfulConf, readRootfulService, rootfulConfRefusal, rootfulConfRetries, rootfulUnreadList, runtimeObservationKey } from "./runtime-observations.mjs";
36
50
 
37
51
  import { makeRunContainer } from "./run-container.mjs";
52
+ import { resolveProviderCredential } from "./env-allowlist.mjs";
38
53
  import { makeSecretsResolver } from "./secrets.mjs";
39
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
54
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
40
55
  import { makeRunMirror } from "./run-mirror.mjs";
41
56
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
42
57
  import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
@@ -46,6 +61,44 @@ import { makeStallGuard } from "./scheduler-stall-guard.mjs";
46
61
  /** How long boot will wait for `docker image inspect` before shipping without a digest. */
47
62
  const BOOT_IMAGE_TIMEOUT_MS = 5_000;
48
63
 
64
+ /**
65
+ * How long boot waits for the daemon facts read (issues #341 and #345): the image read's 5 s, unless the floor asks for a
66
+ * word only a DAEMON observation earns (`isolation`, `mountSet`), when it waits the facts read's own bound plus 2 s. A
67
+ * busy host's `docker info` is the slow read, and a floor this boot met from the table before must not exit 1 on every
68
+ * restart of a healthy daemon. Exported for its test.
69
+ */
70
+ export function bootFactsBoundMs({ backends, backendFloor }) {
71
+ const needsDaemon = unobservedFloor(backends, backendFloor, { [DOCKER_ENDPOINT_LOCAL]: true }).length > 0;
72
+ return needsDaemon ? DAEMON_FACTS_TIMEOUT_MS + 2_000 : BOOT_IMAGE_TIMEOUT_MS;
73
+ }
74
+
75
+ /**
76
+ * How long after a failed forge-auth re-resolve before another job is allowed to try again (issue #316).
77
+ *
78
+ * The in-flight promise dedupes concurrent callers; this bounds sequential ones. Thirty seconds is short
79
+ * enough that a forge coming back is picked up within one job of it, and long enough that a worker
80
+ * draining a thousand-job backlog against a forge that is still down opens tens of identity calls rather
81
+ * than a thousand -- each of which can otherwise sit on undici's 300-second header timeout with the queue
82
+ * waiting behind it.
83
+ */
84
+ const AUTH_RETRY_COOLDOWN_MS = 30_000;
85
+
86
+ /**
87
+ * How long a job will wait for a forge-auth re-resolve before giving up on it.
88
+ *
89
+ * This is the bound that keeps the re-resolve off the critical path, and without it the feature is a
90
+ * wedge rather than a repair. `mintToken` is awaited inside `runJob`, none of the four identity
91
+ * resolvers passes an `AbortSignal`, and undici's default `headersTimeout` is FIVE MINUTES -- so a forge
92
+ * that accepts the connection and then answers nothing (a load balancer draining, a firewall that drops
93
+ * rather than rejects) would hold every job for that long, with BullMQ renewing the lock the whole time
94
+ * so nothing ever stalls out. At `PI_CONCURRENCY=1` that is the queue stopped, with no error line.
95
+ *
96
+ * Ten seconds is far longer than a healthy `GET /user` and far shorter than anything an operator would
97
+ * call a hang. Losing the race does not cancel the underlying call: it is left to settle into the
98
+ * cooldown, so the next job still benefits from whatever it eventually learns.
99
+ */
100
+ const AUTH_RESOLVE_TIMEOUT_MS = 10_000;
101
+
49
102
  /**
50
103
  * How long a fleet-wide scope claim lives. `JOB_TIMEOUT_MS` plus slack: a container cannot outlive that
51
104
  * ceiling, so the claim cannot expire underneath a live job -- which is what makes a refresh unnecessary
@@ -68,97 +121,40 @@ const WORKER_VERSION = (() => {
68
121
  }
69
122
  })();
70
123
 
124
+ // `makeWatchCloser` lives in its own module since issue #301 (the receiver's triggers watch registers
125
+ // the same handle); re-exported here so every existing importer keeps its address.
126
+ export { makeWatchCloser } from "./watch-closer.mjs";
127
+
71
128
  /**
72
- * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
73
- * the new worker starts draining. A leaked container keeps spending, so it must go before any new
74
- * job launches.
75
- *
76
- * It runs `docker ps` / `docker rm -f` ONLY. It never inspects a container's exit code, never touches
77
- * the queue, and never re-enqueues -- queue and retry state belong to Redis, not to docker
78
- * (INT-RUNNER-EXIT-CODE-PROTOCOL / CONST-RETRY-INFRA-ONLY). It logs container names only (no PII).
79
- *
80
- * It assumes ONE worker per docker daemon: a co-located second worker's boot would remove the first's
81
- * in-flight `pi-job-*` container. That is the accepted v1 shape (DES-CONCURRENCY-3, single worker per
82
- * host).
83
- *
84
- * `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
85
- * `reaper_skipped`, and boot continues to the worker.
86
- */
87
- /**
88
- * The stop handle every live-edit watch below hands back, so `startWorker` can register it in the same
89
- * `extraClosers` list that already closes the queues and the host registry (`index.mjs` -> shutdown).
90
- *
91
- * A WATCH NOTHING CAN CLOSE IS NOT A DETAIL (issue #295). `watch(dir, cb).unref?.()` retained nothing, so
92
- * the watch outlived the worker that armed it, and the reload it later fired ran through THAT boot's
93
- * `log` closure: that boot's injected `write`, stamped with that boot's `workerName`. One process running
94
- * one worker, that is a rounding error at exit. One process running forty boots, which is what a test file
95
- * is, and a worker that shut down two tests ago writes into a live worker's capture under a host that is
96
- * not running -- `every log line carries the host` went red in CI reading `runnervmejwal` where it
97
- * asserted `mac-mini-1`.
98
- *
99
- * UNREF'D IS NOT CLEANED UP, and that difference is what hid this across three features. `unref` says only
100
- * that a handle will not hold the event loop open; the watch stays armed either way.
101
- * `INT-HOST-REGISTRY-CONTRACT` states the same distinction from the opposite side, where a bound's timer
102
- * is deliberately NOT unref'd because an unref'd timer does not fire when the hung command is the last
103
- * thing holding the loop.
129
+ * Race a read against a fuse, and CLEAN THE FUSE UP whichever side wins (issue #300). The fuse is
130
+ * unref'd, deliberately: it exists so a wedged docker daemon cannot hold boot, and it must never itself
131
+ * hold the process. But unref'd is not cleaned up (#295's lesson, both halves): when the read won, the
132
+ * old inline race left its five-second timer armed for the full term. The LOSING read stays pending --
133
+ * a wedged `docker inspect` has no cancel -- which is the read-is-a-nicety posture the call site
134
+ * documents, unchanged here.
104
135
  *
105
- * Three properties, each one a way the shutdown breaks without it:
136
+ * NOT FOR BARE CONTEXTS, and the boundary is the fuse's own unref: awaited when nothing else holds the
137
+ * event loop, the fuse never fires and node exits mid-await (measured, exit 13). In this boot the shared
138
+ * redis client and the workers hold the loop, so the fallback always arrives; a caller with an empty
139
+ * loop needs a ref'd timer and a different trade.
106
140
  *
107
- * - It CLOSES the FSWatcher, which is the leak itself.
108
- * - It CANCELS the debounce the watcher already armed. Closing a watcher does not cancel a `setTimeout`
109
- * the callback already set, and only the watcher was ever unref'd -- the 150ms timer never was. In a
110
- * real worker that costs nothing, because the shutdown ends in `process.exit(0)` either way; it is the
111
- * harness, where the loop is left to drain on its own, that the stray timer reaches.
112
- * - Its `close()` NEVER THROWS for the handles its three callers build, and is idempotent -- by NULLING
113
- * what it closed rather than by an early return, which would be a guard with nothing behind it. Node's
114
- * own `FSWatcher.close()` is already both (measured: a second close returns early and neither throws),
115
- * but this closer must not
116
- * INHERIT that guarantee, it must MAKE it: the shutdown loop in `index.mjs` cannot catch a SYNCHRONOUS
117
- * throw from a closer, and the comment there carries the argument. The swallow is
118
- * `makeHostRegistry.close`'s posture rather than a new one.
119
- *
120
- * A watch that was never created -- the `catch` arm of each function below, a platform without `fs.watch`
121
- * -- still gets a closer, so registration is unconditional and the list's shape never depends on the
122
- * platform. That is why all three return from OUTSIDE their try/catch.
123
- *
124
- * WHAT IT CANNOT DO, because the list above would otherwise read as complete: cancel a reload that has
125
- * ALREADY started. `reloadSchedules` is async and awaits a Valkey round trip, so a debounce that fired
126
- * just before the close is still running after it -- and the watchers stay armed for the whole drain
127
- * ahead of the closer loop, not merely 150ms. That reload cannot be recalled, so what is gated instead is
128
- * its VOICE: `reloadLog` below goes quiet once `closed` is set, and every reload is handed that instead
129
- * of the boot's own `log`, which is what the
130
- * issue actually asks for -- a stopped worker writes no line. The reload's own Valkey work may still be
131
- * cut off mid-flight by the queue closing beside it, leaving a scheduler set the next boot's reconcile
132
- * repairs; that race predates this change and is not narrowed by it.
133
- *
134
- * EXPORTED for the reason `reloadScopedLimits` is: none of the three properties is observable through a
135
- * real `fs.watch` without racing the filesystem, and a guarantee the shutdown rests on deserves a
136
- * deterministic pin rather than a sleep.
141
+ * EXPORTED for the reason `makeWatchCloser` is: the cleared-fuse property is not observable through a
142
+ * full boot without racing every other timer the boot arms, and a guarantee the shutdown story rests on
143
+ * deserves a deterministic pin rather than a census.
137
144
  */
138
- export function makeWatchCloser(handles, log) {
139
- return {
140
- // The reload's voice, and the reason this factory is handed the boot's `log` rather than only its
141
- // handles. A reload already in flight cannot be recalled, so what the close gates is what it can
142
- // still SAY: after `closed`, a line from this watch would carry the host of a worker that has
143
- // stopped, which is the bleed the issue is about. The arming lines keep the real `log` -- they run
144
- // before any close.
145
- reloadLog: (event, fields) => {
146
- if (!handles.closed) log(event, fields);
147
- },
148
- close() {
149
- // `closed` FIRST, before anything is torn down: it is what gates `reloadLog` above and the watch
150
- // callback below, so a callback or a reload landing mid-close is already silenced.
151
- handles.closed = true;
152
- clearTimeout(handles.timer);
153
- handles.timer = null;
154
- try {
155
- handles.watcher?.close();
156
- } catch {
157
- // A close that failed has already stopped mattering, and a THROW here rejects the shutdown.
158
- }
159
- handles.watcher = null;
160
- },
161
- };
145
+ export async function settleWithin(promise, ms, fallback) {
146
+ let timer = null;
147
+ try {
148
+ return await Promise.race([
149
+ promise,
150
+ new Promise((resolve) => {
151
+ timer = setTimeout(() => resolve(fallback), ms);
152
+ timer.unref?.();
153
+ }),
154
+ ]);
155
+ } finally {
156
+ clearTimeout(timer);
157
+ }
162
158
  }
163
159
 
164
160
  /**
@@ -168,24 +164,36 @@ export function makeWatchCloser(handles, log) {
168
164
  * schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
169
165
  * `startWorker` registers so the watch dies with the worker that armed it (issue #295).
170
166
  */
171
- function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
167
+ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot) {
172
168
  const path = config.triggersFile;
173
169
  const dir = dirname(path) || ".";
174
170
  const file = basename(path);
175
171
  const handles = { watcher: null, timer: null, closed: false };
176
172
  const closer = makeWatchCloser(handles, log);
173
+ const readFile = () => readFileSync(path, "utf8");
174
+ // The baseline is what the BOOT LOAD read, handed in: the reconcile above and everything between it and
175
+ // this arm -- the endpoint probe, forge auth, the reaper, Valkey -- is the window an edit is lost in
176
+ // (issue #386). A baseline taken here would measure the arming instead.
177
+ readBeforeArming(handles, readFile, atBoot);
177
178
  try {
178
179
  handles.watcher = watch(dir, (_event, changed) => {
179
180
  if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
180
181
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
181
182
  clearTimeout(handles.timer);
182
- handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), 150);
183
+ handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), WATCH_DEBOUNCE_MS);
183
184
  });
184
185
  handles.watcher.unref?.();
185
186
  log("triggers_watching", { path });
186
187
  } catch (err) {
187
188
  log("triggers_watch_unavailable", { reason: err?.message });
188
189
  }
190
+ // ONLY WHEN THE BYTES MOVED, because this one costs a Valkey round trip and a reconcile: a quiet boot
191
+ // must not pay for the race it did not lose, and must not log a second `schedules` line saying nothing
192
+ // changed.
193
+ if (changedWhileArming(handles, readFile)) {
194
+ log("triggers_reread_after_arming", { path });
195
+ void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet });
196
+ }
189
197
  return closer;
190
198
  }
191
199
 
@@ -195,7 +203,7 @@ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
195
203
  * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
196
204
  * FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
197
205
  */
198
- function watchPauseWindowsFile(config, ref, log) {
206
+ function watchPauseWindowsFile(config, ref, log, atBoot) {
199
207
  const path = config.pauseWindowsFile;
200
208
  const dir = dirname(path) || ".";
201
209
  const file = basename(path);
@@ -209,18 +217,26 @@ function watchPauseWindowsFile(config, ref, log) {
209
217
  closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
210
218
  }
211
219
  };
220
+ const readFile = () => readFileSync(path, "utf8");
221
+ readBeforeArming(handles, readFile, atBoot); // the BOOT LOAD's own bytes; see watch-closer (issue #386)
212
222
  try {
213
223
  handles.watcher = watch(dir, (_event, changed) => {
214
224
  if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
215
225
  if (changed && changed !== file) return;
216
226
  clearTimeout(handles.timer);
217
- handles.timer = setTimeout(reload, 150);
227
+ handles.timer = setTimeout(reload, WATCH_DEBOUNCE_MS);
218
228
  });
219
229
  handles.watcher.unref?.();
220
230
  log("pause_windows_watching", { path });
221
231
  } catch (err) {
222
232
  log("pause_windows_watch_unavailable", { reason: err?.message });
223
233
  }
234
+ // SAID, like the triggers watch says it: without a line of its own an operator cannot tell a boot-race
235
+ // reload from an ordinary one, and this is the only reload that happens with nobody editing.
236
+ if (changedWhileArming(handles, readFile)) {
237
+ log("pause_windows_reread_after_arming", { path });
238
+ reload();
239
+ }
224
240
  return closer;
225
241
  }
226
242
 
@@ -244,24 +260,30 @@ export function reloadScopedLimits(config, ref, log) {
244
260
  * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
245
261
  * unref'd and the returned closer stops the watch with the worker (issue #295).
246
262
  */
247
- function watchScopedLimitsFile(config, ref, log) {
263
+ function watchScopedLimitsFile(config, ref, log, atBoot) {
248
264
  const path = config.scopedLimitsFile;
249
265
  const dir = dirname(path) || ".";
250
266
  const file = basename(path);
251
267
  const handles = { watcher: null, timer: null, closed: false };
252
268
  const closer = makeWatchCloser(handles, log);
269
+ const readFile = () => readFileSync(path, "utf8");
270
+ readBeforeArming(handles, readFile, atBoot); // the BOOT LOAD's own bytes; see watch-closer (issue #386)
253
271
  try {
254
272
  handles.watcher = watch(dir, (_event, changed) => {
255
273
  if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
256
274
  if (changed && changed !== file) return;
257
275
  clearTimeout(handles.timer);
258
- handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), 150);
276
+ handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
259
277
  });
260
278
  handles.watcher.unref?.();
261
279
  log("scoped_limits_watching", { path });
262
280
  } catch (err) {
263
281
  log("scoped_limits_watch_unavailable", { reason: err?.message });
264
282
  }
283
+ if (changedWhileArming(handles, readFile)) {
284
+ log("scoped_limits_reread_after_arming", { path });
285
+ reloadScopedLimits(config, ref, closer.reloadLog);
286
+ }
265
287
  return closer;
266
288
  }
267
289
 
@@ -292,34 +314,118 @@ export async function startWorker(
292
314
  // tests in `start-wiring.test.mjs` were reported as not existing at all -- no name, no count, exit 0 --
293
315
  // because this function's log line went through the same channel the runner needed (issue #266).
294
316
  write = (chunk) => process.stdout.write(chunk),
317
+ // Issue #354: the configuration loader, a seam for ONE reason: a test venue that is not in the backend table (an
318
+ // injected `extraBackends` bundle) cannot be written through the real loader, which refuses a name it does not know.
319
+ // The table's own venues (`local`, and `podman` since part 2) go through the real loader.
320
+ loadConfig: loadConfigFn = loadConfig,
295
321
  makeAuth = makeGitHubAuth,
296
322
  makeHost = makeGitHubHost,
297
323
  createWorkerFn = createWorker,
298
324
  makeReaper: makeReaperFn = makeReaper,
299
325
  makeBackendRegistry: makeBackendRegistryFn = makeBackendRegistry,
300
- // Additional backend bundles, in registration order after `local`. The deployment still decides which
301
- // are BLESSED (PI_BACKENDS) and which is default; this only says which exist.
326
+ // Additional backend bundles, in registration order after `local` (which is built only while blessed,
327
+ // issue #354). The deployment still decides which are BLESSED (PI_BACKENDS) and which is default; this only
328
+ // says which exist.
302
329
  extraBackends = [],
303
330
  makeLogSink: makeLogSinkFn = makeLogSink,
304
331
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
305
332
  makeRunMirror: makeRunMirrorFn = makeRunMirror,
306
333
  makeLogReaper: makeLogReaperFn = makeLogReaper,
307
334
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
335
+ makeSandboxNetworkSweeper: makeSandboxNetworkSweeperFn = makeSandboxNetworkSweeper,
336
+ // Issue #429: the one call that asks a runtime which sandboxes are open, seamed so a wiring test can drive the real
337
+ // sandbox reaper against a real retention root without spawning docker or podman.
338
+ listRunningSandboxes: listRunningSandboxesFn = listRunningSandboxes,
339
+ makeRetentionSweep: makeRetentionSweepFn = makeRetentionSweep,
308
340
  makeRunContainer: makeRunContainerFn = makeRunContainer,
309
341
  makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
310
342
  makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
311
343
  makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
312
344
  makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
313
345
  makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
346
+ // Which docker endpoint this host's CLI resolves (issue #278). A seam because the real one spawns the
347
+ // docker CLI, and a wiring test must decide what it answers.
348
+ resolveDockerEndpoint: resolveDockerEndpointFn = makeDockerEndpointResolver(),
349
+ // Issue #341: the one `docker info` the job-user decision reads, and the process facts it reads beside it.
350
+ // Seams for the same reason as the endpoint: a wiring test decides what the daemon and the process say.
351
+ readDaemonFacts: readDaemonFactsFn = makeDaemonFactsReader(),
352
+ // `home` (issue #354) is the account whose rootless Podman runs the podman venue's jobs: its own mounts.conf and
353
+ // containers.conf are read from there. Absent in a test's identity, it falls to the observation's own default.
354
+ jobUserIdentity = { platform: process.platform, release: osRelease(), euid: process.geteuid?.(), egid: process.getegid?.(), home: homedir() },
355
+ // Issue #345: the host files the runtime-mounts observation reads (Podman's mounts.conf and containers.conf). A seam so
356
+ // a wiring test never reads this machine's /etc.
357
+ observationFs = { statSync, readFileSync, readdirSync },
358
+ // Issue #448: `systemctl show podman.service`, read only where a local job runs on rootful Podman's Docker API on this
359
+ // host. A seam so a wiring test decides what the unit says, and never asks this machine's systemd.
360
+ readPodmanService: readPodmanServiceFn = makePodmanServiceReader(),
361
+ // Issue #354: the podman venue's three boot collaborators, each a seam for the endpoint's reason: the real ones spawn
362
+ // `podman`, and a wiring test must decide what `podman info` says, what the reaper lists and what the bundle is
363
+ // built from. Constructing the default reader spawns nothing; only a call does, and only while `podman` is blessed.
364
+ readPodmanInfo: readPodmanInfoFn = makePodmanInfoReader(),
365
+ makePodmanReaper: makePodmanReaperFn = makePodmanReaper,
366
+ makePodmanBackend: makePodmanBackendFn = makePodmanBackend,
314
367
  makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
315
368
  makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
316
369
  makeForgejoAuth: makeForgejoAuthFn = makeForgejoAuth,
317
370
  makeForgejoHost: makeForgejoHostFn = makeForgejoHost,
318
371
  makeAzureAuth: makeAzureAuthFn = makeAzureAuth,
319
372
  makeAzureHost: makeAzureHostFn = makeAzureHost,
373
+ // The clock the forge-auth re-resolve cooldown reads. Injected because a test that asserts a window
374
+ // beside a subject built on the default `Date.now` is a fuse: it passes until the wall clock drifts
375
+ // past that window, then fails in CI on a tree nobody touched (issue #284).
376
+ now = () => Date.now(),
377
+ // Issue #476: how the boot waits out a rootless network keeper that is only too young. A seam so a wiring test
378
+ // proves the wait and its bound without spending real seconds.
379
+ sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
380
+ // How long a job waits for a forge-auth re-resolve. A seam rather than a constant because the
381
+ // property under test is that the bound EXISTS, and a test that proved it by waiting ten real
382
+ // seconds would be paid for on every run for the life of the file.
383
+ authResolveTimeoutMs = AUTH_RESOLVE_TIMEOUT_MS,
384
+ // Issue #464: the jobs dir, created and checked as this account's at boot. A seam so a wiring test decides it.
385
+ ensureJobsDir: ensureJobsDirFn = ensureJobsDir,
386
+ // Issue #464: the same rule for the two durable stores, which default into that root on an account with no home.
387
+ ensureUnderAccountRoot: ensureUnderAccountRootFn = ensureUnderAccountRoot,
388
+ // Issue #464 (gate round 1): the sandbox dir, refused at boot when another account owns it, as the jobs dir is.
389
+ ensureSandboxDir: ensureSandboxDirFn = ensureSandboxDir,
390
+ // Issue #464 (gate round 2): whose Valkey VALKEY_URL reaches, judged here, where the connection is made, and the
391
+ // literal address every Valkey client of this worker then connects to. A seam: the real one reads /proc, probes
392
+ // this host's addresses and resolves the name, none of which belongs in a wiring test.
393
+ judgeValkey = defaultJudgeValkey,
320
394
  } = {},
321
395
  ) {
322
- const config = loadConfig(env);
396
+ const config = loadConfigFn(env);
397
+ // Issue #354: whether this deployment blesses the local adapter (its name is `DEFAULT_BACKEND`, the word
398
+ // `makeLocalBackend` stamps on its bundle). `local` stopped being mandatory in `PI_BACKENDS`, and every read
399
+ // below that asks THIS HOST'S DOCKER CLI something is a read about that one venue: its endpoint, its daemon's
400
+ // facts, its image, its reaper, its sandboxes. With `local` unblessed none of them runs, so a host without
401
+ // Docker never spawns `docker` at boot for a venue it will never use, and a floor naming an observation only
402
+ // `local` makes cannot hold such a worker at exit 1 forever. Everything is byte-identical while it is blessed.
403
+ const localBlessed = config.backends.includes(DEFAULT_BACKEND);
404
+ // The same split for the native podman venue (issue #354, `DES-PODMAN-NATIVE-ROOTLESS-BACKEND`): its `podman info`
405
+ // read, its reaper and its bundle exist only while it is blessed, so a deployment that never chose it spawns no
406
+ // `podman` and pays nothing for it being in the table.
407
+ const podmanBlessed = config.backends.includes(PODMAN_BACKEND);
408
+ // Issue #354: an ABSENT `observationPreflight` admits every job, which is the right answer only for a venue whose
409
+ // words hold without observing anything. A venue whose table entry names an `observedBy` and carries no preflight
410
+ // would have those words hold with nothing looking, and a floor naming one would pass on capability alone: the
411
+ // believed-in control `unarmedFloor` was written against. The registry cannot read the table, so it is refused
412
+ // here, and HERE rather than beside the registry: this runs before anything is spawned or connected, so the refusal
413
+ // leaves nothing open behind it. Only the extra bundles need it; `local`'s is built below carrying its preflight.
414
+ for (const bundle of extraBackends) {
415
+ // A string name only: `backendFor(undefined)` answers with the table's default, and a nameless bundle is the
416
+ // registry's refusal to make, with its own message.
417
+ const gated = typeof bundle?.name === "string" ? Object.keys(backendFor(bundle.name)?.observedBy ?? {}) : [];
418
+ if (gated.length > 0 && typeof bundle?.observationPreflight !== "function") {
419
+ throw new Error(`backend ${JSON.stringify(bundle.name)} declares ${gated.join(", ")} as held only while observed, and carries no observationPreflight to observe them`);
420
+ }
421
+ }
422
+ // Issue #464: the jobs dir is this account's, or the worker does not start. Before it built anything, and a
423
+ // `configError` (exit 2, never restarted) where another account owns it, since a restart meets the same owner: the
424
+ // shared `<tmp>/pi-dispatch/jobs` default let one account's jobs dir fail every other account's jobs with EACCES
425
+ // while the worker ran on. A failed mkdir is the fs error it is (exit 1).
426
+ ensureJobsDirFn(config.jobsDir, { env });
427
+ ensureSandboxDirFn(config.sandboxDir, { env });
428
+ for (const store of [config.logsDir, dirname(config.settingsFile)]) ensureUnderAccountRootFn(store, { env });
323
429
  // `host` sits AFTER the spread, so it is authoritative rather than overridable (issue #57). No call
324
430
  // site can know better than this closure which process wrote a line, and one that passed `host` would
325
431
  // be lying by construction -- verified: none does. This is also why the stamp lives ONLY here. Every
@@ -328,6 +434,22 @@ export async function startWorker(
328
434
  // while one added inside this closure cannot reach them.
329
435
  const log = (event, fields = {}) => write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
330
436
 
437
+ // Issue #464 (gate round 2): the owner rule where the connection is made. `service install`, `up` and doctor judge
438
+ // the Valkey too, but a VALKEY_URL edited after they refused it, a 0.0.0.0 URL, or another account publishing on ::1
439
+ // after the install each reached another account's queue through an unjudged worker (measured on Fedora 44). So the
440
+ // worker judges it itself, before it contacts Valkey at all, with the same function (`resolveWorkerValkey`), and
441
+ // every client below connects to the judged literal address, never to a name resolved again later. A refusal is a
442
+ // configError (exit 2, not restarted into the same answer); nothing answering is a plain error (exit 1, restarted).
443
+ const valkey = await judgeValkey({ url: config.valkeyUrl, venues: { localUsed: localBlessed, podmanUsed: podmanBlessed }, env });
444
+ for (const note of valkey.notes ?? []) log("valkey_note", { note });
445
+ if (valkey.pinned) log("valkey_pinned", { address: valkey.pinned.address, port: valkey.pinned.port, heldBy: valkey.pinned.heldBy });
446
+ // Every client below connects through connection.mjs' JudgedConnector, which judges the (pinned) address again on
447
+ // each connect, as this boot judged it: root refused as the boot decided it, PI_VALKEY_SHARED from the deployment .env.
448
+ const valkeyContext = workerValkeyContext(valkey, env);
449
+ // Issue #468: whether this worker sends a password, and never the password: the only thing a log line may say of it.
450
+ log("valkey_password", { set: Boolean(valkeyPasswordFor(valkey.url, valkeyContext).password) });
451
+ const valkeyConn = (opts = {}) => parseConnection(valkey.url, { ...opts, servername: valkey.servername, context: valkeyContext });
452
+
331
453
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
332
454
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
333
455
  // than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
@@ -336,16 +458,172 @@ export async function startWorker(
336
458
  // scheduled, and a `const` frozen at boot would make it publish the pre-edit set forever -- so two
337
459
  // hosts would see each other's fingerprint oscillate on the beat period, refusing or agreeing
338
460
  // depending on which half of a beat a reload happened to land in.
339
- const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared }) };
461
+ // THE BOOT LOADS' OWN READS are the baselines for the three watches' boot-race checks (issue #386),
462
+ // captured through the seam each loader already has rather than re-read where the watch arms. The window
463
+ // this race lives in is between these lines and the arming a thousand lines below -- the endpoint probe,
464
+ // forge auth, the reaper, Valkey -- not the microseconds around the arming itself, which is what a first
465
+ // attempt measured. `null` where a file is not configured, which reads as "nothing to compare".
466
+ const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null };
467
+ const recording = (into, path) => ({
468
+ readFileSync: (file, enc) => {
469
+ const text = readFileSync(file, enc);
470
+ if (file === path) atBoot[into] = text;
471
+ return text;
472
+ },
473
+ });
474
+ const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared, ...recording("triggers", config.triggersFile) }) };
340
475
 
341
476
  // REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
342
477
  // Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
343
478
  // pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
344
- const pauseWindows = { current: loadPauseWindows(config) };
479
+ const pauseWindows = { current: loadPauseWindows(config, recording("pauseWindows", config.pauseWindowsFile)) };
345
480
 
346
481
  // Issue #242: same posture for the scoped-limits file -- fail-loud with the operator present, mutable
347
482
  // ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
348
- const scopedLimits = { current: loadScopedLimits(config) };
483
+ const scopedLimits = { current: loadScopedLimits(config, recording("scopedLimits", config.scopedLimitsFile)) };
484
+
485
+ // Issue #278: WHICH DOCKER DAEMON the job containers' credentials will travel to. Asked of the CLI at boot,
486
+ // AFTER the free file validations above (a wedged CLI costs up to its bound, and must not delay them) and
487
+ // BEFORE forge auth, the reaper, Valkey and the worker -- so a refusal here stops a process that built
488
+ // nothing. Without a floor naming credentialTransit a redirect is words only: pointing the worker at a
489
+ // daemon is the operator's call. With one asking for `enforced`, it refuses: a floor is not met by capability
490
+ // alone. A TRANSIENT failure to ask (a timeout, a spawn out of resources) throws untagged, exit 1, so the
491
+ // supervisor retries; everything else is a config error, exit 2.
492
+ //
493
+ // Issue #354: judged for `local` ALONE, not for every blessed venue. These are observations of this host's docker
494
+ // CLI and its daemon, which mean nothing for another venue's words: a venue with its own `observedBy` brings its own
495
+ // boot read (the podman venue's is below, judged for `podman` alone the same way).
496
+ const bootEndpoint = localBlessed ? await resolveDockerEndpointFn() : null;
497
+ if (bootEndpoint) {
498
+ logDockerEndpoint(log, bootEndpoint);
499
+ const [endpointRefusal] = observationRefusals({
500
+ backends: [DEFAULT_BACKEND],
501
+ backendFloor: config.backendFloor,
502
+ observations: { [DOCKER_ENDPOINT_LOCAL]: bootEndpoint.local === true },
503
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(bootEndpoint) },
504
+ // The daemon has not been read yet; the runtime observations are checked once it has (issue #345).
505
+ only: [DOCKER_ENDPOINT_LOCAL],
506
+ });
507
+ if (endpointRefusal) throw bootEndpoint.local === null && bootEndpoint.transient ? new Error(endpointRefusal) : configError(endpointRefusal);
508
+ }
509
+ // The per-job read (below, `observationPreflight`) logs only when the answer CHANGES from the last one, so a
510
+ // deliberate, standing redirect writes one line at boot rather than one per job.
511
+ let endpointSeen = bootEndpoint ? dockerEndpointState(bootEndpoint) : null;
512
+
513
+ // Issue #341: WHO job containers run as on this daemon (`DES-JOB-USER-INFERRED-READ-BACK-ON-REQUEST`). Decided
514
+ // from facts, never a probe container, cached per endpoint state. Bounded like the boot image read, because a
515
+ // wedged daemon must not hang boot. Only an IDENTITY verdict refuses at boot, and only when `local` is the
516
+ // default venue: rootless, userns-remap, a root worker and Docker Desktop on Linux cannot run any local job here.
517
+ // An unknown answer (a daemon still starting) boots, so a unit with RestartPreventExitStatus=2 is never stranded
518
+ // by one; so does `runtime-unreadable`, which a later job re-reads.
519
+ const resolveJobUser = makeJobUserResolver({ readFacts: readDaemonFactsFn, ...jobUserIdentity });
520
+ // Issue #345: a floor that needs a DAEMON observation waits the facts read's own bound (plus a margin), not the image
521
+ // read's 5 s: a busy host's `docker info` is the slow read, and a floor this boot met from the table before would
522
+ // otherwise exit 1 on every restart of a healthy daemon.
523
+ // Issue #354: `null` throughout with `local` unblessed. There is no local job to decide a user for, so nothing is
524
+ // read, nothing is said, and the boot line carries `jobUser: null` rather than a decision about a daemon nobody asked.
525
+ const bootJobUser = bootEndpoint
526
+ ? await settleWithin(
527
+ resolveJobUser({ endpoint: bootEndpoint, key: endpointSeen }).catch(() => null),
528
+ bootFactsBoundMs(config),
529
+ null,
530
+ )
531
+ : null;
532
+ const bootDecision = bootEndpoint ? (bootJobUser?.decision ?? { mode: "unknown", user: null, cause: null, reason: "boot-read-timeout" }) : null;
533
+ if (bootDecision) log("job_user", { mode: bootDecision.mode, user: bootDecision.user, cause: bootDecision.cause, reason: bootDecision.reason });
534
+ // Said again whenever a job's decision differs from the last one said, so a boot that read `unknown` (a daemon still
535
+ // starting) and a later answer that refuses every job are never separated by silence.
536
+ let jobUserSaid = bootDecision ? jobUserLogKey(bootDecision) : null;
537
+
538
+ // Issue #345: the RUNTIME observations, from that same facts read and this host's files, checked against the floor
539
+ // the way the endpoint was above: `isolation` holds only while the daemon is observed applying a container's bounds,
540
+ // `mountSet` only while the runtime is observed adding no mounts of its own. Without a floor naming either, this is
541
+ // words (worker_started, and doctor). A refusal resting only on a read that did not answer (a daemon still starting,
542
+ // the boot bound) throws untagged, exit 1, so the supervisor retries; one resting on an answer is a config error.
543
+ // Issue #354: for `local` alone, like the endpoint above, and not at all with `local` unblessed. This is the site that
544
+ // would otherwise hold a worker without `local` at exit 1 forever: a floor naming `isolation` against observations
545
+ // nobody made reads as unanswered, which is the transient arm, which the supervisor retries without end.
546
+ // Issue #448: `systemctl show podman.service`, read once here where the local daemon is rootful Podman on this host, and
547
+ // handed to both the mounts observation and the containers.conf refusal below, so they judge one answer.
548
+ // The deletion rule's memory (gate round 1 of PR #473): which chain files this worker saw while one service start ran.
549
+ const rootfulMemory = makeRootfulMemory();
550
+ const bootUnit = bootEndpoint && bootJobUser?.daemon ? await readRootfulService({ endpoint: bootEndpoint, daemon: bootJobUser.daemon, readService: readPodmanServiceFn }) : undefined;
551
+ const bootObserved = bootEndpoint ? observeHost({ endpoint: bootEndpoint, daemon: bootJobUser?.daemon ?? { answered: false, reason: "boot-read-timeout", transient: true }, fs: observationFs, unit: bootUnit, env, memory: rootfulMemory }) : null;
552
+ if (bootObserved) {
553
+ const bootObservedArgs = { backends: [DEFAULT_BACKEND], backendFloor: config.backendFloor, observations: bootObserved.observations, evidence: bootObserved.evidence };
554
+ const [runtimeRefusal] = observationRefusals(bootObservedArgs);
555
+ if (runtimeRefusal) throw observationRefusalIsTransient(bootObservedArgs) ? new Error(runtimeRefusal) : configError(runtimeRefusal);
556
+ }
557
+ let runtimeObservedSaid = bootObserved ? runtimeObservationKey(bootObserved) : null;
558
+
559
+ const bootRefusal = jobUserBootRefusal(bootDecision, config.defaultBackend);
560
+ if (bootRefusal) throw configError(bootRefusal);
561
+ // Issue #448: where a local job runs on rootful Podman's Docker API service on this host, the containers.conf that
562
+ // service reads (the system chain, root's own, the unit's CONTAINERS_CONF) and whether the running service started
563
+ // before it last changed. A VENUE refusal, as the podman venue's #428 one is and for its reason: with egress off no
564
+ // declared property covers what `env` or `annotations` add to a job. After the identity, whose fix comes first, and
565
+ // only while `local` is the default venue; merely blessed, each local job is refused and the default venue's run.
566
+ // Tagged (exit 2) since a restart reads the same bytes, except what heals by itself (exit 1): a read that failed for a
567
+ // moment, a running service older than its containers.conf, a change time ahead of the clock. What the worker's account
568
+ // cannot read under root's own config home is said once here, never refused on; any other unreadable part refuses.
569
+ const bootRootful = bootEndpoint && bootJobUser?.daemon ? await observeRootfulConf({ endpoint: bootEndpoint, daemon: bootJobUser.daemon, fs: observationFs, readService: readPodmanServiceFn, env, unit: bootUnit, memory: rootfulMemory }) : null;
570
+ let rootfulUnreadSaid = "";
571
+ const sayRootfulUnread = (rootful) => {
572
+ const said = rootfulUnreadList(rootful?.unread);
573
+ if (said === rootfulUnreadSaid) return;
574
+ rootfulUnreadSaid = said;
575
+ if (said !== "") log("local_podman_conf_unread", { unread: said });
576
+ };
577
+ sayRootfulUnread(bootRootful);
578
+ const localConfBoot = localConfBootRefusal(bootDecision, config.defaultBackend, bootRootful);
579
+ if (localConfBoot) throw localConfBoot.transient ? new Error(localConfBoot.message) : configError(localConfBoot.message);
580
+
581
+ // Issue #354: the podman venue's ONE facts read, `podman info --format json`, at boot, where local's reads are and for
582
+ // their reasons: free, before forge auth, the reaper and any Valkey client (the Valkey owner judgement above only reads
583
+ // sockets and probes addresses), so a refusal here stops a process that built nothing.
584
+ // Wrapped once in `cachedPodmanInfo` and the SAME wrapper is handed to the bundle below, so an answer read here is the
585
+ // one the first job is decided from rather than a second spawn. Bounded twice: the reader's own timeout (docker info's
586
+ // 15 s, reused) and this fuse two seconds past it, because a spawn that never settles has no timeout to fire.
587
+ const podmanInfo = podmanBlessed ? cachedPodmanInfo(readPodmanInfoFn) : null;
588
+ const bootPodmanRead = podmanInfo
589
+ ? await settleWithin(
590
+ Promise.resolve()
591
+ .then(() => podmanInfo())
592
+ .catch(() => ({ answered: false, reason: "spawn-failed", transient: true })),
593
+ PODMAN_INFO_TIMEOUT_MS + 2_000,
594
+ { answered: false, reason: "boot-read-timeout", transient: true },
595
+ )
596
+ : null;
597
+ const bootPodmanDecision = bootPodmanRead ? decidePodmanJobUser({ platform: jobUserIdentity.platform, euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, read: bootPodmanRead }) : null;
598
+ // The IDENTITY verdict first, before the observations, and unlike `local`'s order on purpose: a rootful or remote
599
+ // Podman also fails the observations (the mounts check reads a rootless user's files, the service one its remoteness),
600
+ // and a floor refusal naming `podmanAddsNoMounts` would send the operator after a mounts.conf when the fix is the
601
+ // account Podman runs as. Only when `podman` is the DEFAULT venue, as `jobUserBootRefusal` for `local`: with it merely
602
+ // blessed, the jobs that name it are refused one by one and the default venue's still run.
603
+ const podmanRefusal = podmanBootRefusal(bootPodmanDecision, config.defaultBackend);
604
+ if (podmanRefusal) throw configError(podmanRefusal);
605
+ // Issue #428: the account's containers.conf, next, under the same rule (only while `podman` is the default venue; merely
606
+ // blessed, each podman job is refused and the default venue's still run). A VENUE refusal rather than a floor
607
+ // observation: with egress off no declared property covers what a job reaches on the host, so a floor would never
608
+ // fire on the deployments at risk. Read from this host's files with no daemon, so it is determinate whatever the info
609
+ // read said, and tagged (exit 2): a restart reads the same bytes. After the identity, whose fix comes first. The one
610
+ // exception is a read that failed for a moment (out of descriptors, an I/O error), which is untagged (exit 1) so the
611
+ // supervisor restarts it, as an unanswered observation is.
612
+ const podmanConfBoot = podmanConfBootRefusal(bootPodmanDecision, config.defaultBackend, { fs: observationFs, home: jobUserIdentity.home, env, euid: jobUserIdentity.euid, runRoot: bootPodmanRead?.answered === true && bootPodmanRead.info ? (bootPodmanRead.info.runRoot ?? null) : undefined });
613
+ if (podmanConfBoot) throw podmanConfBoot.transient ? new Error(podmanConfBoot.message) : configError(podmanConfBoot.message);
614
+ // The podman venue's own observations, judged for `podman` ALONE, as the endpoint and the daemon's are for `local`
615
+ // alone: each venue's words are earned by its own reads, and judging one venue's answers over every blessed venue
616
+ // would read the other's observations as unanswered, which is the transient arm, exit 1 on every restart. Same split
617
+ // as `local`'s: a refusal resting only on a read that did not answer is untagged (retried), one on an answer tagged.
618
+ const bootPodmanObserved = bootPodmanRead ? observePodman({ read: bootPodmanRead, fs: observationFs, home: jobUserIdentity.home, env, euid: jobUserIdentity.euid }) : null;
619
+ // Not judged at all when the venue's identity is already refused (it is then merely blessed, or the refusal above would
620
+ // have stopped the boot): every job naming it is refused by that cause, and a floor refusal here would stop the whole
621
+ // worker, the default venue's jobs included, with a fix (delegate controllers, empty a mounts.conf) that is not the one.
622
+ if (bootPodmanObserved && bootPodmanDecision?.mode !== "unmappable") {
623
+ const podmanObservedArgs = { backends: [PODMAN_BACKEND], backendFloor: config.backendFloor, observations: bootPodmanObserved.observations, evidence: bootPodmanObserved.evidence };
624
+ const [podmanObservationRefusal] = observationRefusals(podmanObservedArgs);
625
+ if (podmanObservationRefusal) throw observationRefusalIsTransient(podmanObservedArgs) ? new Error(podmanObservationRefusal) : configError(podmanObservationRefusal);
626
+ }
349
627
 
350
628
  // The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
351
629
  // Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
@@ -356,24 +634,137 @@ export async function startWorker(
356
634
  // Auth stays BEST-EFFORT per forge, exactly as it was: a local-only deployment has no GitHub
357
635
  // credentials and must still boot and drain cron jobs. The refusal is deferred to the job that needs
358
636
  // the missing credential (the mintToken fallback below), not raised at startup.
637
+ //
638
+ // WHAT CHANGED (issue #316): best-effort used to mean best-effort ONCE. Any throw left `auth` null for
639
+ // the lifetime of the process, so a forge that was merely unreachable during the seconds this loop ran
640
+ // stayed credential-less until somebody restarted the worker, and every job of that kind then hit the
641
+ // `configError` fallback below -- which, since #310, refunds the reserve and posts a public comment
642
+ // telling the issue author that the operator's deployment is misconfigured. A deployment that
643
+ // `doctor` reports as healthy. The boot posture is unchanged for a DETERMINATE failure, which is what
644
+ // the local-only case is (no `gh` on PATH is `ENOENT`); a TRANSIENT one now leaves a re-resolver
645
+ // behind instead of a permanent null.
359
646
  const forges = { github: { auth: null, host: makeHost() } };
360
- try {
361
- forges.github.auth = await makeAuth(config.github);
362
- log("self_identity", { kind: "github", id: forges.github.auth.selfId, source: forges.github.auth.source });
363
- } catch (err) {
364
- log("github_auth_unavailable", { kind: "github", reason: err?.message });
365
- }
647
+ // Per forge kind, what it would take to resolve its auth again: the closure, and the `idOf` its log
648
+ // line needs. Present only while the last attempt failed transiently -- a determinate failure removes
649
+ // it, because retrying a wrong credential is how a deployment pays to be told the same thing twice.
650
+ const authRetries = new Map();
651
+ const authInFlight = new Map();
652
+ // When the last transient attempt happened, so a backlog draining against a forge that is still down
653
+ // does not open one identity round-trip per job. The in-flight promise below dedupes CONCURRENT
654
+ // callers; this bounds SEQUENTIAL ones, which is the shape a PI_CONCURRENCY=1 worker actually has.
655
+ const authCooldownUntil = new Map();
656
+ const authLastError = new Map();
657
+
658
+ /**
659
+ * Boot-time attempt for one forge, and the record of what to do if it failed.
660
+ *
661
+ * `idOf` exists because azure's selfId is an object while the other three are scalars, which is the
662
+ * one place a forge identity does not reduce to a single value.
663
+ */
664
+ const attachAuth = async (kind, make, cfg, idOf = (auth) => auth.selfId) => {
665
+ try {
666
+ forges[kind].auth = await make(cfg);
667
+ log("self_identity", { kind, id: idOf(forges[kind].auth), source: forges[kind].auth.source });
668
+ } catch (err) {
669
+ // The tag is the whole discriminator, and it is now trustworthy at these sites: the identity
670
+ // modules throw untagged for a fetch rejection, a transient status and an unparseable body.
671
+ const transient = err?.piDispatchConfig !== true;
672
+ if (transient) authRetries.set(kind, { resolve: () => make(cfg), idOf });
673
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient });
674
+ }
675
+ };
676
+
677
+ /**
678
+ * The auth for a forge, resolving it now if boot could not and the reason was transient.
679
+ *
680
+ * THREE bounds, because "ask again" is a live network round-trip on the job path and each of them is a
681
+ * different way of asking too often.
682
+ *
683
+ * ONE in-flight promise per forge dedupes CONCURRENT callers, which is what a worker at
684
+ * PI_CONCURRENCY>1 draining a backlog produces. A COOLDOWN bounds sequential ones: without it a
685
+ * PI_CONCURRENCY=1 worker chewing through a backlog against a forge that is still down opens one
686
+ * identity call per job, forever, and each of them can sit on undici's 300-second header timeout with
687
+ * the queue behind it. Inside the cooldown the last error is re-thrown immediately, which is the same
688
+ * verdict at none of the cost. And a DETERMINATE answer retires the re-resolver entirely.
689
+ *
690
+ * The determinate error is RETHROWN rather than turned into `null`. Returning null sent every such job
691
+ * to the generic "configure GITHUB_AUTH_SOURCE" message while the specific reason -- bad credentials,
692
+ * a key that is not PKCS//8 -- was already in hand and went only to the log. The caller decides what to
693
+ * do with it; the point is that it reaches the caller.
694
+ */
695
+ const ensureAuth = async (kind) => {
696
+ if (forges[kind]?.auth) return forges[kind].auth;
697
+ const retry = authRetries.get(kind);
698
+ if (!retry) return null;
699
+ if (!authInFlight.has(kind)) {
700
+ const until = authCooldownUntil.get(kind) ?? 0;
701
+ if (now() < until) throw authLastError.get(kind) ?? transientError(`${kind} auth is still unavailable`);
702
+ const attempt = async () => {
703
+ try {
704
+ const auth = await retry.resolve();
705
+ forges[kind].auth = auth;
706
+ authCooldownUntil.delete(kind);
707
+ authLastError.delete(kind);
708
+ log("self_identity", { kind, id: retry.idOf(auth), source: auth.source });
709
+ return auth;
710
+ } catch (err) {
711
+ authLastError.set(kind, err);
712
+ if (err?.piDispatchConfig === true) {
713
+ authRetries.delete(kind);
714
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient: false });
715
+ } else {
716
+ authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
717
+ }
718
+ throw err;
719
+ }
720
+ };
721
+ // The cleanup is chained OUTSIDE the async body rather than written as its `finally`, and that is
722
+ // not a style choice. An async function runs synchronously up to its first `await`, so a factory
723
+ // that throws SYNCHRONOUSLY runs the whole body -- catch and finally included -- before this
724
+ // `set` ever happens: the delete would find an empty map and the rejected promise would then be
725
+ // installed permanently, leaving that forge dead for the lifetime of the process. Which is the
726
+ // exact defect issue #316 exists to remove, reintroduced by its own fix. A `.finally` callback
727
+ // is always a microtask, so it cannot outrun the `set`, and the identity check makes it safe
728
+ // against a later attempt having already replaced the entry.
729
+ const inflight = attempt().finally(() => {
730
+ if (authInFlight.get(kind) === inflight) authInFlight.delete(kind);
731
+ });
732
+ // A handler, so that a caller losing the timeout race below cannot turn this into an unhandled
733
+ // rejection. Every awaiter still sees the rejection through its own `await`.
734
+ inflight.catch(() => {});
735
+ authInFlight.set(kind, inflight);
736
+ }
737
+ const inflight = authInFlight.get(kind);
738
+ let timer;
739
+ try {
740
+ return await Promise.race([
741
+ inflight,
742
+ new Promise((_resolve, reject) => {
743
+ timer = setTimeout(() => reject(transientError(`${kind} auth did not answer within ${authResolveTimeoutMs}ms`)), authResolveTimeoutMs);
744
+ timer.unref?.();
745
+ }),
746
+ ]);
747
+ } catch (err) {
748
+ // A lost race is a forge that is not answering, which is exactly what the cooldown is for: the
749
+ // in-flight call is still out there and will set it when it settles, but the next job must not
750
+ // queue up behind it in the meantime.
751
+ if (!authLastError.has(kind)) {
752
+ authLastError.set(kind, err);
753
+ authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
754
+ }
755
+ throw err;
756
+ } finally {
757
+ clearTimeout(timer);
758
+ }
759
+ };
760
+
761
+ await attachAuth("github", makeAuth, config.github);
366
762
  // GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
367
763
  // with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
368
764
  // answer than an entry that exists and cannot authenticate.
369
765
  if (config.gitlab) {
370
766
  forges.gitlab = { auth: null, host: makeGitLabHostFn({ apiUrl: config.gitlab.apiUrl }) };
371
- try {
372
- forges.gitlab.auth = await makeGitLabAuthFn(config.gitlab);
373
- log("self_identity", { kind: "gitlab", id: forges.gitlab.auth.selfId, source: forges.gitlab.auth.source });
374
- } catch (err) {
375
- log("gitlab_auth_unavailable", { kind: "gitlab", reason: err?.message });
376
- }
767
+ await attachAuth("gitlab", makeGitLabAuthFn, config.gitlab);
377
768
  }
378
769
  // Forgejo joins on the same best-effort terms. Its auth can fail for one reason the others cannot: a
379
770
  // repository-scoped token cannot call GET /user, so an operator who scoped their token without setting
@@ -381,24 +772,14 @@ export async function startWorker(
381
772
  // credential-less, which refuses its jobs at mint time rather than running them unattributed.
382
773
  if (config.forgejo) {
383
774
  forges.forgejo = { auth: null, host: makeForgejoHostFn({ apiUrl: config.forgejo.apiUrl }) };
384
- try {
385
- forges.forgejo.auth = await makeForgejoAuthFn(config.forgejo);
386
- log("self_identity", { kind: "forgejo", id: forges.forgejo.auth.selfId, source: forges.forgejo.auth.source });
387
- } catch (err) {
388
- log("forgejo_auth_unavailable", { kind: "forgejo", reason: err?.message });
389
- }
775
+ await attachAuth("forgejo", makeForgejoAuthFn, config.forgejo);
390
776
  }
391
777
  // Azure joins on the same terms. Its selfId is an OBJECT (`{ id, email }`) rather than a scalar, because
392
778
  // a pull-request delivery names an actor by GUID and a work item names them only by address -- the one
393
779
  // place a forge's identity does not reduce to a single value.
394
780
  if (config.azure) {
395
781
  forges.azure = { auth: null, host: makeAzureHostFn({ orgUrl: config.azure.orgUrl }) };
396
- try {
397
- forges.azure.auth = await makeAzureAuthFn(config.azure);
398
- log("self_identity", { kind: "azure", id: forges.azure.auth.selfId?.id ?? null, source: forges.azure.auth.source });
399
- } catch (err) {
400
- log("azure_auth_unavailable", { kind: "azure", reason: err?.message });
401
- }
782
+ await attachAuth("azure", makeAzureAuthFn, config.azure, (auth) => auth.selfId?.id ?? null);
402
783
  }
403
784
 
404
785
  /** The `{ auth, host }` pair a job's kind names, or `undefined` for a local job (which has no forge). */
@@ -423,40 +804,97 @@ export async function startWorker(
423
804
  // and a forgotten one is INVISIBLE, because `reapAll` is conservative over the reapers it is handed
424
805
  // rather than over the venues that exist. A bundle already carries its own `reap`, so taking it from
425
806
  // there is one place. `local`'s is built here because its bundle cannot exist yet.
426
- backendReaps = { [DEFAULT_BACKEND]: makeReaperFn({ log }), ...Object.fromEntries(extraBackends.map((b) => [b?.name, b?.reap])) };
427
- reaped = (await reapAll(Object.values(backendReaps), { log }))?.reaped === true;
807
+ // Issue #354: `local`'s only while it is blessed, since its bundle is built only then and the registry refuses a
808
+ // boot reaper for a venue it does not hold.
809
+ // Issue #354: podman's the same way and for the same reason, `makeReaper` with `bin: "podman"`, so the sweep lists
810
+ // the store this account's jobs actually ran in. Docker's listing says nothing about it, and the reverse.
811
+ backendReaps = {
812
+ ...(localBlessed ? { [DEFAULT_BACKEND]: makeReaperFn({ log }) } : {}),
813
+ ...(podmanBlessed ? { [PODMAN_BACKEND]: makePodmanReaperFn({ log }) } : {}),
814
+ ...Object.fromEntries(extraBackends.map((b) => [b?.name, b?.reap])),
815
+ };
816
+ const swept = (await reapAll(Object.values(backendReaps), { log }))?.reaped === true;
817
+ // PROVEN FOR THE WHOLE HOST, or not at all (issue #354). `reapAll` is conservative over the reapers it is
818
+ // handed, and two gaps sit outside that. A BLESSED venue with no reaper here is refused by the registry, but
819
+ // only further down, after the scope sweep has already acted on this answer. And a host without `local` can
820
+ // still hold `pi-job-` containers under Docker, from before `local` was dropped from PI_BACKENDS, that no
821
+ // blessed venue's reaper lists: "every blessed venue enumerated" is then true while this host is not shown
822
+ // clean. The scope sweep is an optimisation over the TTL, so declining it costs one TTL of a stale claim, never
823
+ // a slot; claiming it would free slots for containers that may still be running. A venue that can prove the
824
+ // Docker side too (a read-only listing) is what lifts the second gap, and it is not this change.
825
+ reaped = swept && localBlessed && config.backends.every((name) => typeof backendReaps[name] === "function");
826
+ // Its own event, said once at boot: the sweeper's `scope_claims_sweep_skipped` says only that the reap did not
827
+ // prove the host, and this is the one case where every reaper answered and the host is still not proven.
828
+ if (swept && !reaped) log("host_reap_unproven", { reason: localBlessed ? "a blessed backend has no boot reaper" : "local is not blessed, so no reaper lists this host's docker containers" });
428
829
  } catch (err) {
429
- log("reaper_skipped", { reason: err?.message });
830
+ log("reaper_skipped", { reason: scrubCredentials(err?.message) });
430
831
  }
431
832
 
432
833
  // REQ-LOCAL-JOB-VISIBILITY: sweep aged `.log`/`.json` history at boot so the logs directory stays
433
834
  // bounded across restarts. Best-effort with the same double-wrap posture as the container reaper: the
434
835
  // reaper swallows its own fs errors, and this guard keeps any reaper failure from blocking draining.
836
+ // HELD, not discarded: the periodic sweep (issue #292) re-runs this exact closure, so one configuration
837
+ // read serves boot and every tick after it and the two cannot drift. The `let` with an inert default is
838
+ // what keeps a throwing FACTORY from leaving the sweep holding `undefined` -- the guard below promises
839
+ // that no reaper failure blocks draining, and that promise now has to cover construction too.
840
+ let reapLogs = () => {};
435
841
  try {
436
- await makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log })();
842
+ reapLogs = makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log });
843
+ await reapLogs();
437
844
  } catch (err) {
438
- log("log_reaper_skipped", { reason: err?.message });
845
+ log("log_reaper_skipped", { reason: scrubCredentials(err?.message) });
439
846
  }
440
847
 
441
848
  // REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
442
849
  // kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
443
- // retention policy, a different PII class, and one thing neither sibling needs: it asks docker which
444
- // sandboxes are live first, because an operator's shell can outlive a worker restart by design and
850
+ // retention policy, a different PII class, and one thing neither sibling needs: it asks the container runtimes
851
+ // which sandboxes are live first, because an operator's shell can outlive a worker restart by design and
445
852
  // deleting a bind mount underneath it is a confusing failure with a boring cause. Same double-wrap.
853
+ // Held for the periodic sweep, same as the log reaper above. Safe to re-run on a timer without any
854
+ // state carried between calls: `makeSandboxReaper` declares `running` INSIDE its returned function and
855
+ // re-issues `listRunning` every call, so a docker outage costs one interval of retention overshoot
856
+ // rather than latching the sweep off until the next boot.
857
+ let reapSandboxes = async () => {};
446
858
  try {
447
- await makeSandboxReaperFn({
859
+ // Issue #429, as corrected by its review: which sandboxes are open is asked PER RETAINED RUN, of the runtime that
860
+ // run's manifest records, never of the runtimes this worker blesses. The opener's blessing is the OPENER's
861
+ // `PI_BACKENDS` and routinely differs from the worker's (`OQ-038`), so a blessed-list listing let a podman-only
862
+ // worker delete a local run's directory under a docker shell an operator had just opened. A runtime that cannot
863
+ // answer holds its own runs this pass and nothing else, so a stale docker CLI does not stop a podman sweep, and a
864
+ // run it cannot place (an unreadable manifest, a venue with no launcher) is asked of every runtime present.
865
+ // `blessed` only adds runtimes to the network sweep; a host that blesses neither and retains nothing from either
866
+ // spawns neither CLI.
867
+ // The store a podman run recorded is compared with THIS worker's podman store (review round 2): the cached boot
868
+ // read when podman is blessed, else one read through the same seam, asked only when a podman run recorded one.
869
+ const readPodmanStore = async () => {
870
+ const read = await (podmanInfo ?? readPodmanInfoFn)();
871
+ return read?.answered === true ? (read.info?.graphRoot ?? null) : null;
872
+ };
873
+ const watch = makeSandboxRuntimeWatch({ sandboxDir: config.sandboxDir, blessed: config.backends, list: listRunningSandboxesFn, readPodmanStore, makeSweeper: makeSandboxNetworkSweeperFn, proxy: config.egressProxy, log });
874
+ reapSandboxes = makeSandboxReaperFn({
448
875
  sandboxDir: config.sandboxDir,
449
876
  retentionHours: config.sandboxRetentionHours,
450
- listRunning: listRunningSandboxes,
877
+ listRunning: watch.listRunning,
878
+ // Issue #337: the session networks a died-mid-session process left, one sweeper per runtime present (issue
879
+ // #429), each listing and removing in its own runtime. Unconditional on `PI_EGRESS`, on the boot reaper's own
880
+ // precedent (it lists `pi-job-` networks whatever the posture) and for a sharper reason: a deployment that has
881
+ // turned the policy OFF is exactly where the leftovers are guaranteed dead, since nothing is making new ones.
882
+ sweepNetworks: watch.sweepNetworks,
883
+ // Issue #446, gate round 1: each expired run's own runtime is asked once more right before it is renamed aside.
884
+ isOpen: watch.isOpen,
451
885
  log,
452
- })();
886
+ });
887
+ await reapSandboxes();
453
888
  } catch (err) {
454
- log("sandbox_reaper_skipped", { reason: err?.message });
889
+ log("sandbox_reaper_skipped", { reason: scrubCredentials(err?.message) });
455
890
  }
456
891
 
457
892
  // One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
458
893
  // hoisted out of the createWorkerFn arg object.
459
- const redis = makeRedisClient(config.valkeyUrl);
894
+ const redis = makeRedisClient(valkey.url, { servername: valkey.servername, context: valkeyContext });
895
+ // Its errors as one message-only line (PR #475's review, round 2), like every Queue and Worker's: without a listener
896
+ // ioredis printed a stack per reconnect attempt.
897
+ onValkeyError(redis, "shared client");
460
898
 
461
899
  // This host's own stale scope claims, gated on the reaper having having enumerated: the
462
900
  // reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
@@ -468,14 +906,14 @@ export async function startWorker(
468
906
  if (config.workerNameDeclared)
469
907
  await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current.map((r) => ({ concurrent: r.concurrent, hash: scopeKeyPrefix(r.scope).slice("budget:s:".length) })), log })({ reaped });
470
908
  } catch (err) {
471
- log("scope_claims_sweep_skipped", { reason: err?.message });
909
+ log("scope_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
472
910
  }
473
911
 
474
912
  // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
475
913
  // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
476
914
  // Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
477
915
  // drains it after the worker.
478
- const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
916
+ const runtimeQueue = makeQueue(valkeyConn());
479
917
 
480
918
  // THE HOST QUEUE (issue #57): work only this machine can do, because the folder lives here.
481
919
  //
@@ -491,7 +929,7 @@ export async function startWorker(
491
929
  // live triggers-file edit lands on the same queue the boot reconcile used; otherwise the shared
492
930
  // runtime queue, exactly as before. Registered as an extraCloser only when it is a NEW handle --
493
931
  // closing `runtimeQueue` twice would be closing another owner's connection.
494
- const cronQueue = hostQueue ? makeQueue(parseConnection(config.valkeyUrl), { name: hostQueue }) : runtimeQueue;
932
+ const cronQueue = hostQueue ? makeQueue(valkeyConn(), { name: hostQueue }) : runtimeQueue;
495
933
 
496
934
  // REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
497
935
  // captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
@@ -510,15 +948,19 @@ export async function startWorker(
510
948
  maxAgeDays: config.sessionMaxAgeDays,
511
949
  maxResumeChain: config.sessionMaxResumeChain,
512
950
  maxContextPct: config.sessionMaxContextPct,
951
+ // The venue a transcript is stamped with and gated on (#277), resolved with the registry's own default.
952
+ defaultBackend: config.defaultBackend,
513
953
  log,
514
954
  });
515
955
  // Boot sweep, beside the log reaper and for the same reason it is beside rather than inside it: these
516
- // files have a different retention policy and a different PII class. The gate that actually matters is
517
- // the age check at OPEN -- a worker that never restarts would otherwise resume forever (OQ-007).
956
+ // files have a different retention policy and a different PII class. Until issue #292 the age check at
957
+ // OPEN was carrying this alone, because a worker that never restarts never re-swept; the periodic sweep
958
+ // now covers the DISK half and the open gate covers the INPUT half, which is the one that matters
959
+ // (OQ-007, RESOLVED).
518
960
  try {
519
961
  sessionStore.reapSessions();
520
962
  } catch (err) {
521
- log("session_reaper_skipped", { reason: err?.message });
963
+ log("session_reaper_skipped", { reason: scrubCredentials(err?.message) });
522
964
  }
523
965
  // The one-shot file path (issue #231): PI_TRIGGERS_FILE, else ./triggers.json against this process's
524
966
  // cwd -- doctor's own fallback, chosen for doctor's own reason ("the two must read the same file"),
@@ -535,7 +977,9 @@ export async function startWorker(
535
977
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
536
978
  // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
537
979
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
538
- const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName });
980
+ // The default venue rides the same way and for the same reason (#277): it is the value the registry
981
+ // below is built with, so the record resolves a job's venue exactly as dispatch does.
982
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend });
539
983
  writeRecord(record);
540
984
  // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
541
985
  // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
@@ -647,16 +1091,102 @@ export async function startWorker(
647
1091
  // ONE preflight instance, constructed once and shared: `start-wiring.test.mjs` pins that, and the
648
1092
  // reason is the module's own -- the tag the preflight checked has to be the tag `docker run` is
649
1093
  // handed, and two constructions are two chances for that to stop being true.
650
- const imagePreflight = makeImagePreflightFn({ image: config.jobImage });
1094
+ // Issue #354: built only while `local` is blessed, since it IS local's (`docker image inspect`). Without `local` the
1095
+ // boot read below asks the default venue's own preflight instead, the one its jobs will be gated on.
1096
+ const imagePreflight = localBlessed ? makeImagePreflightFn({ image: config.jobImage }) : null;
1097
+ // Issue #354: the podman bundle, built only while `podman` is blessed, from the SAME deployment inputs local's
1098
+ // runContainer is handed below (one image, one overlay, one forward list, one egress posture: a venue must not quietly
1099
+ // run a different job than the one configured), plus what only this venue reads: the cached `podman info` the boot
1100
+ // read above already asked, the floor its per-job observation preflight judges, the boot-built reaper, and the
1101
+ // process's identity and host files, through the same seams local's decisions read. Built HERE rather than beside
1102
+ // local's, because the boot image read just below asks the default venue's own preflight when that venue is podman.
1103
+ const podmanBackend = podmanBlessed
1104
+ ? makePodmanBackendFn({
1105
+ image: config.jobImage,
1106
+ hostEnv: env,
1107
+ egress: config.egress,
1108
+ egressProxy: config.egressProxy,
1109
+ openJobLog,
1110
+ globalPiDir: config.globalPiDir,
1111
+ allowGlobalExtensions: config.allowGlobalExtensions,
1112
+ packagePaths: getPackagePaths,
1113
+ forwardEnv: config.forwardEnv,
1114
+ authFromPi: config.authFromPi,
1115
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
1116
+ backendFloor: config.backendFloor,
1117
+ reap: backendReaps[PODMAN_BACKEND],
1118
+ readInfo: podmanInfo,
1119
+ platform: jobUserIdentity.platform,
1120
+ euid: jobUserIdentity.euid,
1121
+ egid: jobUserIdentity.egid,
1122
+ fs: observationFs,
1123
+ home: jobUserIdentity.home,
1124
+ env,
1125
+ log,
1126
+ })
1127
+ : null;
1128
+ // Issue #458 (PR #463 round 2): on Podman 4.x with egress armed, the keeper read ONCE at boot through the bundle's own
1129
+ // preflight, bounded, and said as its own event with the whole sentence. A deployment that upgraded without re-running
1130
+ // `service install` or `up` has no keeper, and would otherwise learn it only from retried jobs. Only a warning: the
1131
+ // per-job preflight is the gate: once the keeper holds, the next job runs (under a proxy up longer than the grace,
1132
+ // 15 s, after the proxy's restart, which the job's retry message and doctor both ask for).
1133
+ if (podmanBackend && config.egress) {
1134
+ const readBootKeeper = () =>
1135
+ settleWithin(
1136
+ Promise.resolve()
1137
+ .then(() => podmanBackend.egressPreflight({}))
1138
+ .catch(() => ({})),
1139
+ BOOT_IMAGE_TIMEOUT_MS * 4,
1140
+ {},
1141
+ );
1142
+ let bootKeeper = await readBootKeeper();
1143
+ // Issue #476: a keeper whose only fault is its age is waited out, once and bounded (at most the minimum age plus
1144
+ // the margin), then judged again. A stack started together reads it under a second old (measured 0.5 to 0.8 s on
1145
+ // 4.9.3), and warning then was wrong every time; a keeper still young after the wait restarted in it, and is said.
1146
+ if (bootKeeper?.young && typeof bootKeeper.young === "object") {
1147
+ const waitMs = Math.min(Number.isFinite(bootKeeper.young.waitMs) ? bootKeeper.young.waitMs : 0, NETNS_KEEPER_MIN_AGE_MS + NETNS_KEEPER_YOUNG_MARGIN_MS);
1148
+ // Said as information, not a warning: the one line that shows a joint start was waited out, and for how long.
1149
+ log("netns_keeper_young_at_boot", { keeperAgeMs: bootKeeper.young.ageMs ?? null, waitMs });
1150
+ await sleep(Math.max(0, waitMs));
1151
+ bootKeeper = await readBootKeeper();
1152
+ }
1153
+ // Its own sentence (PR #463 round 3): the job-shaped one says "this job is retried", and at boot there is no job.
1154
+ if (typeof bootKeeper?.keeper === "string") log("netns_keeper_not_holding_at_boot", { reason: bootKeeper.keeperAtBoot ?? bootKeeper.keeper });
1155
+ }
1156
+ // Every bundle this boot built beside local's, in registration order: the podman one first, then the injected ones.
1157
+ const builtBackends = [...(podmanBackend ? [podmanBackend] : []), ...extraBackends];
651
1158
  // BOUNDED, because `.catch()` cannot rescue a promise that never settles: `runDocker` resolves only on
652
1159
  // the child's `close` or `error` and has no timeout of its own, so a wedged daemon would hang boot
653
1160
  // here. This read is a nicety -- a digest for the boot line and the registry -- and a nicety may
654
1161
  // never be able to stop a worker starting. The per-JOB preflight keeps its unbounded wait, where a
655
1162
  // wedged daemon is the job's problem and the 30-minute job timeout already covers it.
656
- const bootImage = await Promise.race([
657
- imagePreflight({}).catch(() => ({})),
658
- new Promise((resolve) => setTimeout(() => resolve({}), BOOT_IMAGE_TIMEOUT_MS).unref?.()),
659
- ]);
1163
+ // Without `local` (issue #354) the default venue's own preflight, under the same bound and the same swallow: a
1164
+ // nicety that throws synchronously must not stop a boot either, hence the `then`. None at all reads as no digest.
1165
+ const defaultVenueImageRead = () => {
1166
+ const read = builtBackends.find((b) => b?.name === config.defaultBackend)?.imagePreflight;
1167
+ return Promise.resolve()
1168
+ .then(() => (typeof read === "function" ? read({}) : {}))
1169
+ .then((answer) => answer ?? {})
1170
+ .catch(() => ({}));
1171
+ };
1172
+ const bootImage = await settleWithin(imagePreflight ? imagePreflight({}).catch(() => ({})) : defaultVenueImageRead(), BOOT_IMAGE_TIMEOUT_MS, {});
1173
+ // Issue #341: say it at boot, not only as a refusal on every job. The deployment's default image cannot run as
1174
+ // this worker's uid on this daemon, so every local job that uses it will be refused pre-spend.
1175
+ if (bootDecision?.mode === "worker" && bootImage.ok) {
1176
+ const planned = resolveImageUser(bootDecision, { capabilities: bootImage.capabilities ?? [], euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, socket: bootJobUser?.socket ?? null });
1177
+ if (planned.refused === "job-image-any-uid-unsupported") log("job_image_any_uid_unsupported", { image: config.jobImage });
1178
+ else if (planned.refused) log("job_user_group_refused", { cause: planned.cause });
1179
+ if (planned.user && config.forwardEnv.includes("HOME")) log("forward_env_home_overridden", { reason: "HOME is set beside --user" });
1180
+ }
1181
+ // Issue #354: the same boot sentence for a podman DEFAULT venue, whose image was read from this account's own store
1182
+ // just above. Only as the default: with `local` the default, `bootImage` is docker's copy of the tag, which says
1183
+ // nothing about the one in the podman store. `backend` names the venue, since `local`'s lines carry none.
1184
+ if (config.defaultBackend === PODMAN_BACKEND && bootPodmanDecision?.mode === "worker" && bootImage.ok) {
1185
+ const planned = resolvePodmanImageUser(bootPodmanDecision, { capabilities: bootImage.capabilities ?? [], euid: jobUserIdentity.euid, egid: jobUserIdentity.egid });
1186
+ if (planned.refused === "job-image-any-uid-unsupported") log("job_image_any_uid_unsupported", { image: config.jobImage, backend: PODMAN_BACKEND });
1187
+ else if (planned.refused) log("job_user_group_refused", { cause: planned.cause, backend: PODMAN_BACKEND });
1188
+ if (planned.user && config.forwardEnv.includes("HOME")) log("forward_env_home_overridden", { reason: "HOME is set beside --user", backend: PODMAN_BACKEND });
1189
+ }
660
1190
  // Resolved once: `Intl` is not free, and this value cannot change without a restart.
661
1191
  const hostTz = Intl.DateTimeFormat().resolvedOptions().timeZone ?? "";
662
1192
  const registry = makeHostRegistryFn({ redis, name: config.workerName, log });
@@ -698,6 +1228,107 @@ export async function startWorker(
698
1228
  });
699
1229
 
700
1230
 
1231
+ // `local`'s two optional preflights (issue #354: they were `deps` closures, and are now the bundle's own members,
1232
+ // dispatched per venue by the registry). Bodies unchanged.
1233
+ //
1234
+ // Issue #278: the docker endpoint read AGAIN before each job's spend, because a `docker context use`
1235
+ // after boot redirects every later job, and a preflight that answered once at boot would give a
1236
+ // wrong decision all day (the image preflight is not cached for the same reason). Only for a venue
1237
+ // whose declaration is observation-gated on it, and only a refusal under a floor that needs it.
1238
+ // Issue #345: the runtime observations come from the same per-job read, in the same place: the facts the job user is
1239
+ // decided from (cached per endpoint state, so no second daemon call) and this host's Podman files, re-read per job
1240
+ // because an operator creating the empty mounts.conf override must not need a restart.
1241
+ const localObservationPreflight = async (job) => {
1242
+ const venue = resolveBackendName(job, config.defaultBackend);
1243
+ if (Object.keys(backendFor(venue)?.observedBy ?? {}).length === 0) return { ok: true };
1244
+ const endpoint = await resolveDockerEndpointFn();
1245
+ const state = dockerEndpointState(endpoint);
1246
+ if (state !== endpointSeen) {
1247
+ endpointSeen = state;
1248
+ logDockerEndpoint(log, endpoint, { changed: true });
1249
+ }
1250
+ // The endpoint refusal FIRST, from the CLI's own configuration only, as boot does: a floor that distrusts this
1251
+ // endpoint must not wait on, or send the CLI's own TLS client credentials to, the daemon behind it.
1252
+ const endpointArgs = {
1253
+ backends: [venue],
1254
+ backendFloor: config.backendFloor,
1255
+ observations: { [DOCKER_ENDPOINT_LOCAL]: endpoint.local === true },
1256
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(endpoint) },
1257
+ only: [DOCKER_ENDPOINT_LOCAL],
1258
+ };
1259
+ const [endpointRefusal] = observationRefusals(endpointArgs);
1260
+ if (endpointRefusal) {
1261
+ if (endpoint.local === null && endpoint.transient) return { unavailable: true, reason: endpoint.reason };
1262
+ return { refused: true, message: endpointRefusal, observations: [DOCKER_ENDPOINT_LOCAL] };
1263
+ }
1264
+ const jobUser = await resolveJobUser({ endpoint, key: state });
1265
+ const unit = await readRootfulService({ endpoint, daemon: jobUser.daemon, readService: readPodmanServiceFn });
1266
+ // One read of each host path for this job's two checks (`onceFs`, gate round 3 of PR #473), fresh per job.
1267
+ const jobFs = onceFs(observationFs);
1268
+ const observed = observeHost({ endpoint, daemon: jobUser.daemon, fs: jobFs, unit, env, memory: rootfulMemory });
1269
+ if (runtimeObservationKey(observed) !== runtimeObservedSaid) {
1270
+ runtimeObservedSaid = runtimeObservationKey(observed);
1271
+ log("runtime_observed", { daemonAppliesBounds: observed.observations.daemonAppliesBounds, runtimeAddsNoMounts: observed.observations.runtimeAddsNoMounts, changed: true });
1272
+ }
1273
+ // Issue #448: rootful Podman's containers.conf, per job and before any spend, re-read every time because removing a
1274
+ // key must need no worker restart (only the Podman service's). Handed back as the podman venue's refusal is
1275
+ // (`podmanConfRefused`, `rootful: true`), ahead of the floor: it is what the venue IS on this host, and the processor
1276
+ // returns it before the image preflight. Nothing is read where the daemon is not rootful Podman on this host.
1277
+ const rootful = await observeRootfulConf({ endpoint, daemon: jobUser.daemon, fs: jobFs, readService: readPodmanServiceFn, env, unit, memory: rootfulMemory });
1278
+ sayRootfulUnread(rootful);
1279
+ if (rootful?.refusal) return { ok: true, endpoint, jobUser, podmanConfRefused: rootfulRefused(rootful.refusal) };
1280
+ const args = {
1281
+ backends: [venue],
1282
+ backendFloor: config.backendFloor,
1283
+ observations: observed.observations,
1284
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(endpoint), ...observed.evidence },
1285
+ };
1286
+ const [refusal] = observationRefusals(args);
1287
+ if (!refusal) return { ok: true, endpoint, jobUser };
1288
+ const missed = [...new Set(unobservedFloor(args.backends, args.backendFloor, args.observations).map((m) => m.observedBy))];
1289
+ if (observationRefusalIsTransient(args)) return unavailableFor(observed, missed[0]);
1290
+ return { refused: true, message: refusal, observations: missed };
1291
+ };
1292
+ // Issue #452, gate round 5: the runtime each local job was admitted on, by job id, taken (and forgotten) by that job's
1293
+ // teardown. Bounded: a job refused after its job-user preflight never reaches a teardown, so the oldest entries go first.
1294
+ const admittedRuntimes = new Map();
1295
+ const ADMITTED_RUNTIMES_MAX = 1000;
1296
+ const recordAdmittedRuntime = (job, runtime) => {
1297
+ if (job?.id === undefined || runtime === undefined) return;
1298
+ admittedRuntimes.delete(job.id);
1299
+ admittedRuntimes.set(job.id, runtime);
1300
+ while (admittedRuntimes.size > ADMITTED_RUNTIMES_MAX) admittedRuntimes.delete(admittedRuntimes.keys().next().value);
1301
+ };
1302
+ const takeAdmittedRuntime = (job) => {
1303
+ const runtime = admittedRuntimes.get(job?.id);
1304
+ admittedRuntimes.delete(job?.id);
1305
+ return runtime;
1306
+ };
1307
+ // Issue #341: the job user, for a job on the `local` venue only (its containers are this host's docker
1308
+ // CLI's). The endpoint is the one `observationPreflight` just read, so one job's two decisions agree.
1309
+ const localJobUserPreflight = async (job, { capabilities = [], observed } = {}) => {
1310
+ const venue = resolveBackendName(job, config.defaultBackend);
1311
+ if (venue !== DEFAULT_BACKEND) return { user: null, home: null };
1312
+ const endpoint = observed?.endpoint ?? (await resolveDockerEndpointFn());
1313
+ const admittedOn = observed?.jobUser ?? (await resolveJobUser({ endpoint, key: dockerEndpointState(endpoint) }));
1314
+ const { decision, socket, facts } = admittedOn;
1315
+ // Issue #452, gate round 5: the runtime THIS job is admitted on, recorded per job for its teardown's detach gate. The
1316
+ // resolver's cache is per endpoint state and moves when the endpoint does, so reading it at teardown could hand a
1317
+ // job the answer of another endpoint's daemon.
1318
+ recordAdmittedRuntime(job, admittedOn?.daemon?.answered ? runtimeFromFacts(admittedOn.daemon) : undefined);
1319
+ if (jobUserLogKey(decision) !== jobUserSaid) {
1320
+ jobUserSaid = jobUserLogKey(decision);
1321
+ log("job_user", { mode: decision.mode, user: decision.user, cause: decision.cause, reason: decision.reason });
1322
+ }
1323
+ const chosen = resolveImageUser(decision, { capabilities, euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, socket });
1324
+ // Issue #355: whether this job's own mounts carry `:Z`, from the SAME facts and endpoint the user was decided
1325
+ // from, so one job's two answers cannot come from two reads. Only on a path that runs (a refusal or an
1326
+ // undecidable daemon runs nothing), and only when true: every host this does not apply to keeps the answer
1327
+ // shape, and so the argv, it had before.
1328
+ if (chosen.refused || chosen.unavailable) return chosen;
1329
+ return relabelsPrivateMounts(facts, endpoint, jobUserIdentity.platform) ? { ...chosen, relabel: true } : chosen;
1330
+ };
1331
+
701
1332
  // #227: WHERE this job's container runs. The three functions that decide whether a container may start and
702
1333
  // then start it -- two pre-spend gates and the launcher -- bundled into one value with a completeness
703
1334
  // check, so the set has a name instead of being three unrelated `deps` keys. Byte-identical to passing them individually -- the same three functions reach the same three keys,
@@ -707,67 +1338,105 @@ export async function startWorker(
707
1338
  // Assigned key by key below rather than spread, because the bundle also carries `name` and `declares`,
708
1339
  // and `deps` is the processor's namespace: a spread would put a backend's name into it under a key the
709
1340
  // processor is free to mean something else by.
710
- const localBackend = makeLocalBackend({
711
- // #227. The two the earlier slices deferred, now real seams. `reap` keeps its tri-state: the boot
712
- // sweep below only sweeps this host's scope claims once the reaper has PROVEN this host holds no
713
- // job containers, and an unproven answer must never free a slot.
714
- stopContainer: makeStopContainer(),
715
- reap: backendReaps[DEFAULT_BACKEND],
716
- // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
717
- // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
718
- // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
719
- // Nothing is memoised (see its construction above): `docker image inspect` costs ~tens of ms against a
720
- // container run of minutes, and a cache would be wrong in both directions -- an operator who builds the
721
- // image mid-day would stay refused, one who removes it would stay admitted. Contrast the staged-package
722
- // manifest, correctly read once at boot because it is deploy-time state under a :ro mount.
723
- imagePreflight,
724
- // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
725
- // value, one place, so the gate that checks the proxy and the runner that attaches to its network
726
- // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
727
- // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
728
- // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
729
- egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
730
- runContainer: makeRunContainerFn({
731
- image: config.jobImage,
732
- hostEnv: env,
733
- egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
734
- egressProxy: config.egressProxy,
735
- openJobLog,
736
- globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
737
- allowGlobalExtensions: config.allowGlobalExtensions,
738
- // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
739
- // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
740
- // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
741
- packagePaths: getPackagePaths,
742
- forwardEnv: config.forwardEnv,
743
- authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
744
- // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
745
- // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
746
- // adding one does not widen this signature again.
747
- forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
748
- }),
749
- });
1341
+ //
1342
+ // Issue #354: built ONLY while `local` is blessed. Built unconditionally, as it was while PI_BACKENDS had to include
1343
+ // it, it would be a registered venue this deployment never chose, whose reaper the boot map must then carry and
1344
+ // whose reads spawn `docker` on a host that may have none. The two optional preflights ride on the bundle rather
1345
+ // than through `makeLocalBackend`, whose completeness check treats every member it takes as required.
1346
+ const localBackend = localBlessed
1347
+ ? {
1348
+ ...makeLocalBackend({
1349
+ // #227. The two the earlier slices deferred, now real seams. `reap` keeps its tri-state: the boot
1350
+ // sweep below only sweeps this host's scope claims once the reaper has PROVEN this host holds no
1351
+ // job containers, and an unproven answer must never free a slot.
1352
+ stopContainer: makeStopContainer(),
1353
+ reap: backendReaps[DEFAULT_BACKEND],
1354
+ // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
1355
+ // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
1356
+ // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
1357
+ // Nothing is memoised (see its construction above): `docker image inspect` costs ~tens of ms against a
1358
+ // container run of minutes, and a cache would be wrong in both directions -- an operator who builds the
1359
+ // image mid-day would stay refused, one who removes it would stay admitted. Contrast the staged-package
1360
+ // manifest, correctly read once at boot because it is deploy-time state under a :ro mount.
1361
+ imagePreflight,
1362
+ // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
1363
+ // value, one place, so the gate that checks the proxy and the runner that attaches to its network
1364
+ // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
1365
+ // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
1366
+ // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
1367
+ egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
1368
+ runContainer: makeRunContainerFn({
1369
+ image: config.jobImage,
1370
+ hostEnv: env,
1371
+ egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
1372
+ egressProxy: config.egressProxy,
1373
+ openJobLog,
1374
+ globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
1375
+ allowGlobalExtensions: config.allowGlobalExtensions,
1376
+ // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
1377
+ // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
1378
+ // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
1379
+ packagePaths: getPackagePaths,
1380
+ forwardEnv: config.forwardEnv,
1381
+ authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
1382
+ // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
1383
+ // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
1384
+ // adding one does not widen this signature again.
1385
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
1386
+ // Issue #452, gate round 4: the teardown's detach gate uses the `docker info` facts this job was admitted
1387
+ // on (the job-user resolver's cached answer), never a fresh read, which on Docker Engine read as
1388
+ // `runtime-unreadable` whenever it timed out or failed and leaked the job's network. No cached answer: it
1389
+ // reads. A refused teardown is logged with its token.
1390
+ teardownRuntime: (job) => takeAdmittedRuntime(job),
1391
+ log,
1392
+ }),
1393
+ }),
1394
+ observationPreflight: localObservationPreflight,
1395
+ jobUserPreflight: localJobUserPreflight,
1396
+ }
1397
+ : null;
750
1398
 
751
1399
  // #227. WHICH backend runs which job, and the one place that decides. One bundle today, so every
752
1400
  // resolution returns it -- but the mechanism is real, so `run.backend` stops being a validated label and
753
1401
  // the abort path can reach a venue it did not build. `config.backends[0]` is the deployment's default,
754
1402
  // and the registry refuses a default it does not hold rather than discovering it at the first pickup.
755
- const backends = makeBackendRegistryFn({
756
- // #227. A SEAM, not a literal. `docs/backends.md` tells an adapter author to register their bundle
757
- // here, and until this was injectable that instruction described code nobody could run: the array
758
- // was hard-coded, so a venue could pass the conformance suite, get a table entry and be blessed in
759
- // PI_BACKENDS, and then be refused at boot as blessed-but-unbuilt with nowhere to put it. It is also
760
- // what lets a wiring test prove `startWorker` actually CONNECTS the registry to the processor --
761
- // six mutations reverting that connection survived the whole suite, which is the same shape as the
762
- // bug that shipped: invisible while there is one venue.
763
- bundles: [localBackend, ...extraBackends],
764
- defaultName: config.defaultBackend,
765
- // Cross-checked at boot rather than discovered at the first pickup: a name PI_BACKENDS blesses but
766
- // nothing builds passes both the loader and the pre-spend gate, and a venue with no boot reaper is
767
- // swept by nothing while still reporting the host as proven clean.
768
- blessed: config.backends,
769
- reaps: backendReaps,
770
- });
1403
+ // A refusal here comes AFTER boot opened the Redis client, the runtime and cron queues and the host registry, so all of
1404
+ // them are released before the refusal travels: any one left open keeps the event loop alive, and a worker that refused
1405
+ // to boot would hang instead of exiting (measured against a real Valkey: two sockets held past 30 s).
1406
+ let backends;
1407
+ try {
1408
+ backends = makeBackendRegistryFn({
1409
+ // #227. A SEAM, not a literal. `docs/backends.md` tells an adapter author to register their bundle
1410
+ // here, and until this was injectable that instruction described code nobody could run: the array
1411
+ // was hard-coded, so a venue could pass the conformance suite, get a table entry and be blessed in
1412
+ // PI_BACKENDS, and then be refused at boot as blessed-but-unbuilt with nowhere to put it. It is also
1413
+ // what lets a wiring test prove `startWorker` actually CONNECTS the registry to the processor --
1414
+ // six mutations reverting that connection survived the whole suite, which is the same shape as the
1415
+ // bug that shipped: invisible while there is one venue.
1416
+ bundles: [...(localBackend ? [localBackend] : []), ...builtBackends],
1417
+ defaultName: config.defaultBackend,
1418
+ // Cross-checked at boot rather than discovered at the first pickup: a name PI_BACKENDS blesses but
1419
+ // nothing builds passes both the loader and the pre-spend gate, and a venue with no boot reaper is
1420
+ // swept by nothing while still reporting the host as proven clean.
1421
+ blessed: config.backends,
1422
+ reaps: backendReaps,
1423
+ });
1424
+ } catch (err) {
1425
+ const opened = [registry, runtimeQueue, ...(cronQueue !== runtimeQueue ? [cronQueue] : [])];
1426
+ // Bounded, then forced: a queue whose connection came up closes by sending QUIT and awaiting the reply, and against a
1427
+ // server that stopped answering that wait never ends (measured through a stalling proxy), so the refusal itself would
1428
+ // never travel. Five seconds covers the host registry's own bounded close; whatever is still open is disconnected.
1429
+ await settleWithin(Promise.allSettled(opened.map((handle) => Promise.resolve().then(() => handle?.close?.()))), 5_000);
1430
+ for (const handle of opened) {
1431
+ try {
1432
+ handle?.disconnect?.();
1433
+ } catch {
1434
+ // a handle with nothing left to drop
1435
+ }
1436
+ }
1437
+ redis.disconnect();
1438
+ throw err;
1439
+ }
771
1440
 
772
1441
  // The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
773
1442
  // array rather than the literal it used to be, because up to three of its members do not exist yet: the
@@ -781,8 +1450,81 @@ export async function startWorker(
781
1450
  // issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
782
1451
  const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
783
1452
 
1453
+ // HOISTED out of the deps literal (issue #288): the terminal failed listener below needs the same
1454
+ // adapter the processor gets, and two bodies would drift exactly where drift costs a public comment.
1455
+ const comment = async (job, text) => {
1456
+ // Best-effort: the processor awaits comment() inside its try, so a rejection here would
1457
+ // corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
1458
+ // This adapter NEVER throws.
1459
+ const forge = forgeFor(job);
1460
+ // Same re-resolve as the mint path, and it matters more here: a comment is how a refusal
1461
+ // reaches the person who asked for the job, so a forge whose auth was merely unreachable at
1462
+ // boot must not degrade every later comment to a stdout line nobody is watching.
1463
+ //
1464
+ // A THROWING re-resolve is treated as no auth rather than as a failed comment, which is what
1465
+ // keeps the fallthrough below reachable: the text still lands on stdout instead of being
1466
+ // replaced by a `comment_failed` line that does not carry it. This adapter never throws.
1467
+ let auth = forge?.auth ?? null;
1468
+ if (forge && !auth) {
1469
+ try {
1470
+ auth = await ensureAuth(job.kind);
1471
+ } catch {
1472
+ auth = null;
1473
+ }
1474
+ }
1475
+ if (auth) {
1476
+ try {
1477
+ const token = await auth.mintToken(job);
1478
+ await forge.host.postStatusComment(job, job.target, text, token);
1479
+ } catch (err) {
1480
+ log("comment_failed", { jobId: job?.id, reason: err?.message });
1481
+ }
1482
+ return;
1483
+ }
1484
+ // A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
1485
+ // post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
1486
+ log("comment", { jobId: job?.id, text });
1487
+ };
1488
+
1489
+ // The operator's failure hook (issue #288, INT-ON-FAILURE-HOOK-CONTRACT). Constructed ONLY when the
1490
+ // knob is set: an unset PI_ON_FAILURE builds nothing, spawns nothing, and logs nothing -- the
1491
+ // byte-identical guarantee. Fired from the two terminal listeners below, never from the processor:
1492
+ // a hook fault must not be able to flip an outcome, and the listeners sit outside every try that
1493
+ // decides one.
1494
+ // `hostEnv: env`, not the default process.env -- the #309 rule every spawner in this file follows
1495
+ // (runContainer, resolveSecrets): a subprocess runs with THIS worker's env, identical on the real
1496
+ // path and divergent only under an injected one, which is exactly where the difference would hide.
1497
+ const onFailure = config.onFailure
1498
+ ? makeOnFailure({ command: config.onFailure, timeoutMs: config.onFailureTimeoutMs, host: config.workerName ?? "", hostEnv: env, log })
1499
+ : null;
1500
+ // Which POLICY reasons page the operator. Paid terminals only: worker-abort, runner-policy and every
1501
+ // named runner reason cost a container and ended wrong. The runner's named reasons are SPREAD from
1502
+ // RUNNER_POLICY_REASONS rather than listed, because each is an exit 2 that would have paged as
1503
+ // runner-policy before it got its own label, and a label must not be the thing that silences a page;
1504
+ // provider-auth-refused (issue #437) is the case in point, a refusal only the operator can fix. Excluded on purpose: `completed` and every pre-spend refusal (free, and
1505
+ // each already comments -- a delivery storm against a spent cap must not page anyone), and
1506
+ // `operator-cancel`, because the operator initiated it and a push telling them what they just did is
1507
+ // noise with a pager attached.
1508
+ const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", ...RUNNER_POLICY_REASONS]);
1509
+ // The infra-terminal sentence (issue #288). FIXED, never err.message: the message classes that reach
1510
+ // a failedReason carry host paths and library words (the #310 record), and for a local job this text
1511
+ // lands verbatim in the service log through the adapter's stdout fallthrough. The worker log already
1512
+ // holds the truncated reason on the job_failed line beside it.
1513
+ const FAILED_COMMENT = "Failed: an error stopped this job and it will not be retried further. Ask the operator to check the worker log.";
1514
+ // Issue #458 (PR #463 round 2): a job that never started because the podman venue's rootless network keeper did not
1515
+ // hold says so, since the generic line sends an operator to a log that only says the job failed. Fixed text keyed by
1516
+ // the fixed reason token, never the error's own words: a forge comment carries nothing a host read produced.
1517
+ const FAILED_COMMENT_BY_REASON = Object.freeze({
1518
+ // Neutral on the cause (PR #463 round 3): a keeper that is not running and a proxy that must restart after it both land here.
1519
+ [NETNS_KEEPER_NOT_HOLDING]: "Failed before it started: this worker's rootless Podman egress proxy did not pass its pre-start check (its rootless network keeper, pi-dispatch-netns-keeper), so the job was retried and never run. Nothing was spent. Ask the operator to run `pi-dispatch doctor` on the worker for the exact fix.",
1520
+ // Issue #476: a job held on a young keeper that kept restarting while it waited.
1521
+ [NETNS_KEEPER_CRASH_LOOP]: "Failed before it started: this worker's rootless network keeper (pi-dispatch-netns-keeper), which its Podman egress proxy needs, kept restarting while the job waited for it, so the job was retried and never run. Nothing was spent. Ask the operator to run `pi-dispatch doctor` on the worker and read the keeper's log (`journalctl --user -u pi-dispatch-netns-keeper.service`).",
1522
+ // Issue #448 (gate round 2 of PR #473): the hold's own ending, so the comment names the cause and the fix.
1523
+ [PODMAN_RESTART_HOLD_EXPIRED]: "Failed before it started: rootful Podman's service on this worker kept running with a containers.conf older than its files (or a file's change time stayed ahead of the host's clock) for an hour, so the job was held and never run. Nothing was spent. Ask the operator to restart it while no local job runs (`sudo systemctl restart podman.service`), then run the job again.",
1524
+ });
1525
+
784
1526
  const worker = createWorkerFn({
785
- connection: parseConnection(config.valkeyUrl),
1527
+ connection: valkeyConn(),
786
1528
  // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
787
1529
  // not enough to find the runtime holding it once there is more than one venue.
788
1530
  stopContainer: backends.stopContainer,
@@ -847,6 +1589,29 @@ export async function startWorker(
847
1589
  // spending delivery is excused). In the compose topology this check is the once-enforcement
848
1590
  // layer, because the receiver's single-file :ro mount pins a dead inode until restart.
849
1591
  checkOnceSpent: makeCheckOnceSpent({ triggersPath: onceTriggersFile }),
1592
+ // Issue #310. The free provider-credential gate, bound from EXACTLY the values the container builder
1593
+ // is handed, and asserted to be the same by a wiring test. A gate that resolves against different
1594
+ // inputs than the writer is the divergence class this whole cluster of issues is about: it would pass
1595
+ // a job the container then refuses, or refuse one the container would have run. `agentDir` is
1596
+ // defaulted by both, from the same `hostEnv`.
1597
+ //
1598
+ // A processor dep and NOT a backend bundle member: the bundle is a closed set about how a venue runs
1599
+ // a container, and this is a question about the deployment, asked before any venue is chosen.
1600
+ //
1601
+ // A PROBE: whatever it resolves is dropped on the floor. The credential itself is read where it always
1602
+ // was, inside buildContainerEnv, so no live key is ever in scope in the processor.
1603
+ checkProviderCredential: (job) => {
1604
+ try {
1605
+ resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv });
1606
+ return { ok: true };
1607
+ } catch (error) {
1608
+ // Only OUR determinate refusal. Anything else (a bug here, an fs fault the module does not model)
1609
+ // must not become a policy refusal on the operator's issue: it rethrows into runJob's catch, which
1610
+ // classifies it the way it always did.
1611
+ if (error?.piDispatchConfig !== true) throw error;
1612
+ return { ok: false, message: error.message };
1613
+ }
1614
+ },
850
1615
  // Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
851
1616
  // asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
852
1617
  // service below the version floor turns a wait into a paid run nothing can tell from a correct
@@ -878,6 +1643,11 @@ export async function startWorker(
878
1643
  },
879
1644
  imagePreflight: backends.imagePreflight,
880
1645
  egressPreflight: backends.egressPreflight,
1646
+ // Issue #354: both are the VENUE's, dispatched through the registry like the two gates above, so a job on a
1647
+ // venue that carries neither gets the processor's own defaults and never asks this host's docker CLI anything.
1648
+ // `local`'s are the closures built beside its bundle below, unchanged.
1649
+ observationPreflight: backends.observationPreflight,
1650
+ jobUserPreflight: backends.jobUserPreflight,
881
1651
  // Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
882
1652
  // retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
883
1653
  promoteSession: sessionStore.promoteSession,
@@ -893,7 +1663,12 @@ export async function startWorker(
893
1663
  // because the settings file is read at each job start and an operator who declares a profile in
894
1664
  // the panel should not have to restart the worker to use it. A deployment that declares nothing
895
1665
  // spawns nothing at all: the gate only calls this when a trigger is armed.
896
- resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, log }),
1666
+ // `hostEnv` is the env THIS worker was started with, not `process.env` by default (issue #309). It is
1667
+ // the environment the resolver subprocess runs in, and makeRunContainer above is handed the same
1668
+ // `env` for the container it builds. Identical on the real path, where both are process.env; under an
1669
+ // injected env they were not, which is the divergence this file already calls out by name for
1670
+ // sessionsDir. One deployment value, one place, same rule as the profiles below it.
1671
+ resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, hostEnv: env, log }),
897
1672
  // #227. What PI_BACKENDS blessed, so a trigger naming an unblessed venue refuses pre-spend. The
898
1673
  // panel's picker is bounded by the same list, and this is the half that binds: the overlay is
899
1674
  // not the reviewed artifact (DES-PER-TRIGGER-SECRET-PROFILE).
@@ -911,10 +1686,14 @@ export async function startWorker(
911
1686
  neverStartedExits: backends.neverStartedExits,
912
1687
  prepareWorkspace: makePrepareWorkspace({
913
1688
  jobsDir: config.jobsDir,
1689
+ // Issue #464: the boot's own check, asked again before every job against the same env.
1690
+ ensureDir: (dir) => ensureJobsDirFn(dir, { env }),
914
1691
  forgeFor,
915
1692
  // REQ-RESURRECTABLE-SANDBOX: the deployment default, resolved per job against run.image so a
916
1693
  // retained directory records the image that actually ran and a sandbox re-opens that one.
917
1694
  jobImage: config.jobImage,
1695
+ // #277: the venue a retained directory records, which the sandbox refuses by when it is not here.
1696
+ defaultBackend: config.defaultBackend,
918
1697
  preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
919
1698
  // The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
920
1699
  // per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
@@ -927,24 +1706,7 @@ export async function startWorker(
927
1706
  // REQ-RESURRECTABLE-SANDBOX. With the window at 0 this IS the old bare `cleanup`, by the same
928
1707
  // `rm` on the same path -- a deployment that wants no retention keeps today's behaviour exactly.
929
1708
  cleanup: makeCleanup({ sandboxDir: config.sandboxDir, retentionHours: config.sandboxRetentionHours, log }),
930
- comment: async (job, text) => {
931
- // Best-effort: the processor awaits comment() inside its try, so a rejection here would
932
- // corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
933
- // This adapter NEVER throws.
934
- const forge = forgeFor(job);
935
- if (forge?.auth) {
936
- try {
937
- const token = await forge.auth.mintToken(job);
938
- await forge.host.postStatusComment(job, job.target, text, token);
939
- } catch (err) {
940
- log("comment_failed", { jobId: job?.id, reason: err?.message });
941
- }
942
- return;
943
- }
944
- // A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
945
- // post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
946
- log("comment", { jobId: job?.id, text });
947
- },
1709
+ comment,
948
1710
  log,
949
1711
  // Resolved per job so the credential always comes from the job's OWN forge. A job whose forge has
950
1712
  // no working auth refuses here, at mint time, rather than running anonymously -- and the refusal
@@ -956,7 +1718,22 @@ export async function startWorker(
956
1718
  // token comes from must always be something the trigger said.
957
1719
  mintToken: async (job) => {
958
1720
  const kind = job?.kind === "local" ? "github" : job?.kind;
959
- const auth = forges[kind]?.auth;
1721
+ // `ensureAuth` returns the boot-time auth when there is one, and otherwise re-resolves once
1722
+ // if boot's failure was transient (issue #316). It throws the transient failure through, and
1723
+ // that throw is UNTAGGED, so the arm below turns it into the retryable class rather than
1724
+ // letting it reach the processor's config classifier: "the forge was unreachable a moment
1725
+ // ago" is not "this deployment is misconfigured", and only one of those deserves a public
1726
+ // comment saying so.
1727
+ let auth;
1728
+ try {
1729
+ auth = await ensureAuth(kind);
1730
+ } catch (err) {
1731
+ // A determinate re-resolve failure carries the REAL reason (bad credentials, a key that
1732
+ // is not PKCS//8), which is strictly better than the generic message below, so it is
1733
+ // passed straight through rather than collapsed into it.
1734
+ if (err?.piDispatchConfig === true) throw err;
1735
+ throw new InfraRetry(`${kind} auth could not be resolved: ${err?.message ?? "unknown"}`);
1736
+ }
960
1737
  if (auth) return await auth.mintToken(job);
961
1738
  if (kind === "github") {
962
1739
  throw configError("github jobs and cron triggers with run.github require a working GITHUB_AUTH_SOURCE (gh/pat/app)");
@@ -971,105 +1748,350 @@ export async function startWorker(
971
1748
  },
972
1749
  });
973
1750
 
974
- // REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
975
- // where the operator is already looking. This is the local counterpart of the GitHub issue
976
- // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
977
- // what tells a human a run did nothing. The container's own output already streams via
978
- // runContainer's onOutput during the run.
979
- // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
980
- // job-image-missing), never
981
- // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
982
- // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
983
- // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
984
- // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
985
- const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
986
- for (const w of allWorkers) w.on("completed", (job, result) =>
987
- log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
988
- );
989
- for (const w of allWorkers) w.on("failed", (job, err) =>
990
- log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
991
- );
992
-
993
- // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
994
- // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
995
- // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
996
- const onStalled = makeStallGuard({
997
- redis,
998
- threshold: config.schedulerStallMax,
999
- // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
1000
- // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
1001
- removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
1002
- log,
1003
- });
1004
- // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
1005
- // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
1006
- // so every stall threw a TypeError and the money backstop never counted one (issue #267).
1007
- for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
1008
-
1009
- // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
1010
- // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
1011
- // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
1012
- // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
1013
- // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
1014
- // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
1015
- // right place to catch, because it can ask the registry and this cannot.
1016
- const { served, unserved } = servedSchedules(schedules.current);
1017
- for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
1018
-
1019
- if (served.length > 0) {
1020
- // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
1021
- // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
1022
- // config" is correct again by construction and two hosts can no longer prune each other at all. The
1023
- // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
1024
- // disagreement, which no queue split can detect.
1025
- const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
1026
- try {
1027
- const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
1028
- log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
1029
- } finally {
1030
- await rq.close().catch(() => {});
1751
+ // FROM HERE TO THE RETURN, THE WORKER IS LIVE AND THE BOOT CAN STILL REFUSE (issue #299). The
1752
+ // Worker starts consuming at construction, so everything below runs beside a paid drain --
1753
+ // registrations, the stall guard, the boot reconcile (the reachable refusal, reproduced), the
1754
+ // watcher pushes, the retention sweep's construction and start. A refusal that merely threw left
1755
+ // the process printing an error while taking jobs, because `cli.mjs` sets `process.exitCode` and
1756
+ // the live Worker held the loop forever. The catch below covers the REGION, not a list of calls,
1757
+ // so a step added tomorrow is covered the day it is added.
1758
+ try {
1759
+
1760
+ // REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
1761
+ // where the operator is already looking. This is the local counterpart of the GitHub issue
1762
+ // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
1763
+ // what tells a human a run did nothing. The container's own output already streams via
1764
+ // runContainer's onOutput during the run.
1765
+ // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
1766
+ // provider-auth-refused | job-image-missing), never
1767
+ // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
1768
+ // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
1769
+ // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
1770
+ // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
1771
+ const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
1772
+ for (const w of allWorkers) w.on("completed", (job, result) => {
1773
+ log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) });
1774
+ // The hook's POLICY half (issue #288): worker-abort, runner-policy and provider-auth-refused RETURN, so they land here
1775
+ // and never in the failed listener -- a failed-only mount would miss exactly the paid terminals
1776
+ // the feature exists for. Folded into the existing listener body, never a second w.on: the
1777
+ // start-wiring harness records ONE handler per event, and two would race the log line's pin.
1778
+ if (onFailure && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason)) {
1779
+ onFailure({ jobId: job?.id, outcome: "policy", reason: result.reason });
1780
+ }
1781
+ });
1782
+ for (const w of allWorkers) w.on("failed", (job, err) => {
1783
+ log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) });
1784
+ // The one reason whose sentence is the fix itself (issue #458): logged WHOLE, beside the cut line above.
1785
+ if (err?.reason === NETNS_KEEPER_NOT_HOLDING || err?.reason === NETNS_KEEPER_CRASH_LOOP) log("job_failed_netns_keeper", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? "") });
1786
+ // The TERMINAL failed attempt only (issue #288): `finishedOn` is set by BullMQ's own move on the
1787
+ // non-retry branch alone, and the emit follows it, so this guard reads the queue's decision
1788
+ // instead of re-deriving attempts arithmetic that could drift from shouldRetry. A retried
1789
+ // attempt comments nothing (a flaky daemon must not post three comments for one recovery) and
1790
+ // pages nobody; a recovery comments nothing at all. This seam also catches what the processor
1791
+ // never sees: the stall-kill (maxStalledCount 0 fails a crashed worker's job at next pickup)
1792
+ // and the wait-gate rethrow that escapes above the processor's catch.
1793
+ if (job?.finishedOn) {
1794
+ void comment({ ...job.data, id: job.id }, (typeof err?.reason === "string" && Object.hasOwn(FAILED_COMMENT_BY_REASON, err.reason) ? FAILED_COMMENT_BY_REASON[err.reason] : null) ?? FAILED_COMMENT);
1795
+ onFailure?.({ jobId: job?.id, outcome: "failed", reason: typeof err?.reason === "string" ? err.reason : "infra" });
1796
+ }
1797
+ });
1798
+
1799
+ // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
1800
+ // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
1801
+ // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
1802
+ const onStalled = makeStallGuard({
1803
+ redis,
1804
+ threshold: config.schedulerStallMax,
1805
+ // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
1806
+ // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
1807
+ removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
1808
+ log,
1809
+ });
1810
+ // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
1811
+ // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
1812
+ // so every stall threw a TypeError and the money backstop never counted one (issue #267).
1813
+ for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
1814
+
1815
+ // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
1816
+ // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
1817
+ // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
1818
+ // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
1819
+ // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
1820
+ // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
1821
+ // right place to catch, because it can ask the registry and this cannot.
1822
+ const { served, unserved } = servedSchedules(schedules.current);
1823
+ for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
1824
+
1825
+ if (served.length > 0) {
1826
+ // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
1827
+ // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
1828
+ // config" is correct again by construction and two hosts can no longer prune each other at all. The
1829
+ // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
1830
+ // disagreement, which no queue split can detect.
1831
+ const rq = makeQueue(valkeyConn({ failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
1832
+ try {
1833
+ const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
1834
+ log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
1835
+ } finally {
1836
+ await rq.close().catch(() => {});
1837
+ }
1838
+ } else {
1839
+ log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
1031
1840
  }
1032
- } else {
1033
- log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
1034
- }
1035
1841
 
1036
- // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
1037
- // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
1038
- // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1039
- // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
1040
- if (config.triggersFile) {
1041
- extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared));
1042
- }
1842
+ // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
1843
+ // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
1844
+ // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1845
+ // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
1846
+ if (config.triggersFile) {
1847
+ extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers));
1848
+ }
1043
1849
 
1044
- // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
1045
- // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
1046
- if (config.pauseWindowsFile) {
1047
- extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log));
1048
- }
1850
+ // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
1851
+ // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
1852
+ if (config.pauseWindowsFile) {
1853
+ extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log, atBoot.pauseWindows));
1854
+ }
1855
+
1856
+ // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
1857
+ if (config.scopedLimitsFile) {
1858
+ extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log, atBoot.scopedLimits));
1859
+ }
1860
+
1861
+ // issue #292 / OQ-007: re-run the three retention sweeps on a timer, because the supported deployment
1862
+ // is a service that restarts only on failure, so the healthy worker was the one that never re-swept.
1863
+ // Armed HERE, at the end of boot beside the watches: all three closures exist, boot's own sweeps have
1864
+ // long finished, and the first tick lands one full interval later rather than during the schedules
1865
+ // reconcile, which is Redis-destructive work a small test interval would otherwise land inside.
1866
+ //
1867
+ // NOT CONSTRUCTED AT ALL when the knob is 0. That is how "0 is byte-identical to before" is a fact
1868
+ // rather than a claim: with no object there is no timer, no closer and no reachable second call to any
1869
+ // reaper. It is registered in `extraClosers` for issue #295's finding that UNREF'D IS NOT CLEANED UP --
1870
+ // this handle holds an `rmSync`, and a tick landing mid-drain could delete a retained workspace behind a
1871
+ // worker that already reported a clean shutdown.
1872
+ if (config.sweepIntervalHours > 0) {
1873
+ const sweep = makeRetentionSweepFn({
1874
+ reapers: [
1875
+ { name: "log", reap: reapLogs },
1876
+ { name: "sandbox", reap: reapSandboxes },
1877
+ { name: "session", reap: () => sessionStore.reapSessions() },
1878
+ ],
1879
+ intervalMs: config.sweepIntervalHours * 3600000,
1880
+ log,
1881
+ });
1882
+ sweep.start();
1883
+ extraClosers.push(sweep);
1884
+ }
1049
1885
 
1050
- // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
1051
- if (config.scopedLimitsFile) {
1052
- extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log));
1886
+ log("worker_started", {
1887
+ queue: "pi-jobs",
1888
+ // Issue #278: the service's OWN answer to which daemon its jobs go to. `doctor` reads its caller's
1889
+ // shell, and a service's EnvironmentFile or a systemd User= can resolve differently.
1890
+ // Issue #354: all five null with `local` unblessed, where this host's docker CLI was not asked.
1891
+ dockerContext: bootEndpoint ? bootEndpoint.context : null,
1892
+ dockerEndpointLocal: bootEndpoint ? bootEndpoint.local : null,
1893
+ jobUser: bootDecision ? { mode: bootDecision.mode, user: bootDecision.user, cause: bootDecision.cause } : null, // issue #341
1894
+ // Issue #345: whether this daemon is observed applying a container's bounds and adding no mounts of its own; null = not read.
1895
+ daemonAppliesBounds: bootObserved ? bootObserved.observations.daemonAppliesBounds : null,
1896
+ runtimeAddsNoMounts: bootObserved ? bootObserved.observations.runtimeAddsNoMounts : null,
1897
+ // Issue #354: the podman venue's own boot answers, all null with `podman` unblessed (no `podman info` was asked):
1898
+ // the version it reported, whether it is rootless and whose uid its jobs run as, and its three observations.
1899
+ podmanVersion: bootPodmanRead?.answered ? bootPodmanRead.info.version : null,
1900
+ podmanRootless: bootPodmanRead?.answered ? bootPodmanRead.info.rootless : null,
1901
+ podmanJobUser: bootPodmanDecision ? { mode: bootPodmanDecision.mode, user: bootPodmanDecision.user, cause: bootPodmanDecision.cause, reason: bootPodmanDecision.reason } : null,
1902
+ podmanBoundsDelegated: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_BOUNDS_DELEGATED] : null,
1903
+ podmanAddsNoMounts: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_ADDS_NO_MOUNTS] : null,
1904
+ podmanServiceLocal: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_SERVICE_LOCAL] : null,
1905
+ host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
1906
+ imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
1907
+ concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
1908
+ sweepIntervalHours: config.sweepIntervalHours, // 0 = boot-only sweeps, this version's pre-#292 behaviour
1909
+ dailyCap: config.dailyCap,
1910
+ weeklyCap: config.weeklyCap, // null when the weekly window is disabled
1911
+ monthlyCap: config.monthlyCap, // null when the monthly window is disabled
1912
+ softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
1913
+ scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
1914
+ scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
1915
+ image: config.jobImage,
1916
+ valkey: config.valkeyUrl,
1917
+ // Issue #464: the literal address every Valkey client of this worker dials, beside the URL as written; null
1918
+ // where nothing was pinned (another machine's Valkey, dialled by name).
1919
+ valkeyPinned: valkey.pinned ? `${valkey.pinned.address.includes(":") ? `[${valkey.pinned.address}]` : valkey.pinned.address}:${valkey.pinned.port}` : null,
1920
+ logsDir: config.logsDir,
1921
+ settingsFile: config.settingsFile,
1922
+ captureJobLogs: config.captureJobLogs,
1923
+ logRetentionDays: config.logRetentionDays,
1924
+ sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
1925
+ });
1926
+ return worker;
1927
+ } catch (err) {
1928
+ // STOP WHAT WAS BUILT, THEN RETHROW, and both halves carry weight. The stop is the shutdown
1929
+ // minus the exit -- the cancel, the close, the closer drain, the client release -- so nothing is
1930
+ // left holding the loop and the process drains to the refusal's OWN exit code: entryExitCode
1931
+ // turns a tagged configError into EXIT_POLICY 2 and infra into the retryable 1, and a swallow
1932
+ // here would hand a supervisor a clean 0 for a boot that refused. `Promise.resolve`, not an
1933
+ // optional-chained `.catch`: a test double whose recording `stop` is synchronous would otherwise
1934
+ // raise a TypeError OVER the boot's real error, and the exit-code assertion meant to go red
1935
+ // under mutation would go green for the wrong reason. A synchronous throw from `stop` itself
1936
+ // still escapes, exactly as the closer loop documents for `extraClosers` -- a known bound.
1937
+ await Promise.resolve(worker?.stop?.()).catch(() => {});
1938
+ throw err;
1053
1939
  }
1940
+ }
1054
1941
 
1055
- log("worker_started", {
1056
- queue: "pi-jobs",
1057
- host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
1058
- imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
1059
- concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
1060
- dailyCap: config.dailyCap,
1061
- weeklyCap: config.weeklyCap, // null when the weekly window is disabled
1062
- monthlyCap: config.monthlyCap, // null when the monthly window is disabled
1063
- softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
1064
- scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
1065
- scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
1066
- image: config.jobImage,
1067
- valkey: config.valkeyUrl,
1068
- logsDir: config.logsDir,
1069
- settingsFile: config.settingsFile,
1070
- captureJobLogs: config.captureJobLogs,
1071
- logRetentionDays: config.logRetentionDays,
1072
- sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
1942
+ /**
1943
+ * The boot refusal text for a job-user decision, or `null` to boot. Exported so its default-venue branch is pinned on the
1944
+ * predicate as well as end to end (a podman default reaches it since issue #354 part 2, and must not refuse on local's causes).
1945
+ */
1946
+ export function jobUserBootRefusal(decision, defaultBackend) {
1947
+ if (decision?.mode !== "unmappable" || !BOOT_REFUSING_JOB_USER_CAUSES.has(decision.cause)) return null;
1948
+ return defaultBackend === DEFAULT_BACKEND ? jobUserRefusal(decision) : null;
1949
+ }
1950
+
1951
+ /**
1952
+ * The boot refusal text for a podman job-user decision, or `null` to boot (issue #354): an identity cause no podman job
1953
+ * can get past (`PODMAN_BOOT_REFUSING_CAUSES`), and only while `podman` is the DEFAULT venue, `jobUserBootRefusal`'s rule.
1954
+ * Kept apart from that function rather than merged into it, because the two venues' causes share a name
1955
+ * (`worker-is-root`) and not a remedy, and each venue's text comes from its own map.
1956
+ */
1957
+ export function podmanBootRefusal(decision, defaultBackend) {
1958
+ if (defaultBackend !== PODMAN_BACKEND) return null;
1959
+ if (decision?.mode !== "unmappable" || !PODMAN_BOOT_REFUSING_CAUSES.has(decision.cause)) return null;
1960
+ return podmanJobUserRefusal(decision);
1961
+ }
1962
+
1963
+ /**
1964
+ * The boot refusal for a rootful Podman containers.conf that reaches a local job (issue #448), from `observeRootfulConf`'s
1965
+ * answer, or `null`: only while `local` is the default venue, and not when the local job user is already refused (that
1966
+ * refusal, earlier at boot or per job, is the one to fix first). `transient` (thrown untagged, exit 1, so the supervisor
1967
+ * restarts it) is a read that failed for a moment AND, since gate round 1 of PR #473, a service older than its
1968
+ * containers.conf or a change time ahead of the clock: each heals by itself, so none may strand a unit with
1969
+ * `RestartPreventExitStatus=2`. Exported for the doc test, as `podmanConfBootRefusal` is.
1970
+ */
1971
+ export function localConfBootRefusal(decision, defaultBackend, rootful) {
1972
+ if (defaultBackend !== DEFAULT_BACKEND || !rootful?.refusal || decision?.mode === "unmappable") return null;
1973
+ return { message: rootfulConfRefusal(rootful.refusal), transient: rootfulConfRetries(rootful.refusal) };
1974
+ }
1975
+
1976
+ /**
1977
+ * A rootful finding as the processor's `podmanConfRefused` (issue #448): the podman venue's shape, marked `rootful`.
1978
+ * `retry` marks what the processor throws for the queue's retry rather than returns (a moment's read failure, a service
1979
+ * older than its containers.conf, a clock behind a change time), with the evidence its retry message names.
1980
+ */
1981
+ function rootfulRefused(found) {
1982
+ return {
1983
+ reason: found.cause,
1984
+ key: found.key ?? null,
1985
+ message: rootfulConfRefusal(found),
1986
+ rootful: true,
1987
+ ...(found.restart ? { restart: true } : {}),
1988
+ ...(found.skew ? { skew: true } : {}),
1989
+ ...(found.transient ? { transient: true } : {}),
1990
+ ...(rootfulConfRetries(found) ? { retry: true, evidence: found.evidence } : {}),
1991
+ };
1992
+ }
1993
+
1994
+ /**
1995
+ * The boot refusal for a widening containers.conf on the podman venue (issue #428), as `{ message, transient }`, or
1996
+ * `null` to boot: only while `podman` is the DEFAULT venue, `podmanBootRefusal`'s rule, and not when the identity is
1997
+ * already refused (that refusal names the fix that comes first, and with `podman` merely blessed its jobs are refused
1998
+ * one by one anyway). `transient` is a read that failed for a moment, which the caller throws untagged.
1999
+ */
2000
+ export function podmanConfBootRefusal(decision, defaultBackend, files) {
2001
+ if (defaultBackend !== PODMAN_BACKEND || !decision || decision.mode === "unmappable") return null;
2002
+ const widened = podmanConfWidening(files);
2003
+ return widened ? { message: podmanConfRefusal(widened), transient: widened.transient === true } : null;
2004
+ }
2005
+
2006
+ /** What makes two job-user decisions the same for the `job_user` log line. */
2007
+ function jobUserLogKey(decision) {
2008
+ return `${decision?.mode}|${decision?.user ?? ""}|${decision?.cause ?? ""}|${decision?.reason ?? ""}`;
2009
+ }
2010
+
2011
+ /** A one-line summary of an endpoint answer, used only to notice that it changed. */
2012
+ function dockerEndpointState(endpoint) {
2013
+ return `${endpoint.local}|${endpoint.context ?? ""}|${endpoint.endpoint ?? ""}|${endpoint.reason ?? ""}`;
2014
+ }
2015
+
2016
+ /**
2017
+ * What an endpoint answer shows, for a refusal message. The context name is operator config; the endpoint is
2018
+ * the display form, which is the value reduced to `scheme://host` or withheld, never edited (`displayEndpoint`).
2019
+ */
2020
+ function dockerEndpointEvidence(endpoint) {
2021
+ if (endpoint.local === null) return `the docker CLI did not say which endpoint it resolves (${endpoint.reason})`;
2022
+ return `the docker CLI resolves context ${quotedShown(endpoint.context)} to ${endpointShown(endpoint)}${endpoint.local ? ", on this host" : ", which is not shown to be on this host"}`;
2023
+ }
2024
+
2025
+ /** Log an endpoint answer that is not plainly local; a return to local is logged only as a change. */
2026
+ function logDockerEndpoint(log, endpoint, { changed = false } = {}) {
2027
+ if (endpoint.local === false) log("docker_endpoint_not_local", { context: endpoint.context, endpoint: endpoint.endpoint });
2028
+ else if (endpoint.local === null) log("docker_endpoint_unresolved", { reason: endpoint.reason });
2029
+ else if (changed) log("docker_endpoint_local", { context: endpoint.context, endpoint: endpoint.endpoint });
2030
+ }
2031
+
2032
+ /**
2033
+ * Where the worker's clients stand (issue #464): enforced exactly as its boot judgement was (`valkey.enforce`), so a
2034
+ * reconnect is judged by the rule the boot applied, never a looser one; PI_VALKEY_SHARED from the deployment `.env`.
2035
+ */
2036
+ export function workerValkeyContext(valkey, env, { cwd = process.cwd(), readEnv } = {}) {
2037
+ return valkeyClientContext({ env, cwd, rootRefused: valkey?.rootRefused === true, ...(readEnv ? { readEnv } : {}) });
2038
+ }
2039
+
2040
+ /**
2041
+ * The worker's real Valkey judgement (issue #464, gate round 2): `resolveWorkerValkey` with this host's facts. The
2042
+ * deployment `.env` is the one in the working directory (the unit's WorkingDirectory is the deployment folder), read
2043
+ * only for PI_VALKEY_SHARED; nothing else of it is loaded into this process.
2044
+ */
2045
+ async function defaultJudgeValkey({ url, venues, env }) {
2046
+ const envPath = join(process.cwd(), ".env");
2047
+ let envText = null;
2048
+ try {
2049
+ // Bytes (gate round 3): the hardened reader checks what systemd refuses to load before it decodes.
2050
+ envText = readFileSync(envPath);
2051
+ } catch (err) {
2052
+ // No .env here: nothing opts in. Any other failure is said, never read as "no opt-in" (gate round 3).
2053
+ if (err?.code !== "ENOENT") throw configError(`${envPath} could not be read (${err?.code ?? err?.message}), and the worker reads PI_VALKEY_SHARED from it, so it does not start`);
2054
+ }
2055
+ const fs = { readFileSync };
2056
+ const euid = process.geteuid?.();
2057
+ let user = null;
2058
+ try {
2059
+ user = userInfo().username;
2060
+ } catch {
2061
+ // A uid with no passwd entry: its subordinate ranges are read by uid.
2062
+ }
2063
+ const valkey = await resolveWorkerValkey({
2064
+ url,
2065
+ venues,
2066
+ platform: process.platform,
2067
+ env,
2068
+ envText,
2069
+ envPath,
2070
+ probeTcp: probeTcpAddress,
2071
+ lookup: (host, opts) => dnsLookup(host, opts),
2072
+ fs,
2073
+ euid,
2074
+ user,
2075
+ ownerName: (uid) => passwdNameFrom(fs, uid),
2076
+ interfaces: networkInterfaces,
2077
+ subuids: readSubuidRanges({ user, euid, fs }),
2078
+ configError,
1073
2079
  });
1074
- return worker;
2080
+ await refuseValkeyAuth(valkey, env);
2081
+ return valkey;
2082
+ }
2083
+
2084
+ /**
2085
+ * Issue #468: the worker's credential asked once at boot, on the judged address, before anything is built on it. A
2086
+ * Valkey that requires a password this worker does not send (NOAUTH), or refuses the one it sends (WRONGPASS), is a
2087
+ * configError (exit 2, not restarted into the same answer), naming VALKEY_PASSWORD and never its value: left to the
2088
+ * clients, it was an endless stream of NOAUTH errors from a worker that looked alive. Nothing answering is left to the
2089
+ * clients' own retries, as before. Exported for its test; `authState` is the seam.
2090
+ */
2091
+ export async function refuseValkeyAuth(valkey, env, { cwd = process.cwd(), authState = valkeyAuthState } = {}) {
2092
+ const context = workerValkeyContext(valkey, env, { cwd });
2093
+ const { state, error } = await authState(valkey.url, { context, servername: valkey.servername ?? null });
2094
+ // Gate round 2 of PR #478: a database the server does not have (`/16` on a default Valkey) is a refusal, exit 2.
2095
+ if (state === "dbrange") throw configError(error);
2096
+ if (state === "noauth" || state === "wrongpass") throw configError(authRefusalFor(state, valkey.url, context));
1075
2097
  }