@edgehero/pi-dispatch 1.10.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +32 -5
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +387 -17
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1395 -268
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
package/src/start.mjs CHANGED
@@ -1,9 +1,15 @@
1
- import { readFileSync, watch } from "node:fs";
1
+ import { readdirSync, readFileSync, statSync, watch } from "node:fs";
2
+ import { lookup as dnsLookup } from "node:dns/promises";
3
+ import { homedir, networkInterfaces, release as osRelease, userInfo } from "node:os";
2
4
  import { dirname, basename, join } from "node:path";
3
- import { configError, loadConfig } from "./config.mjs";
4
- import { makeRedisClient, parseConnection } from "./connection.mjs";
5
+ import { configError, ensureJobsDir, ensureSandboxDir, ensureUnderAccountRoot, loadConfig } from "./config.mjs";
6
+ import { passwdNameFrom, probeTcpAddress, readSubuidRanges, resolveWorkerValkey } from "./podman-stack.mjs";
7
+ import { valkeyClientContext } from "./valkey-endpoint.mjs";
8
+ import { authRefusalFor, makeRedisClient, onValkeyError, parseConnection, valkeyAuthState, valkeyPasswordFor } from "./connection.mjs";
5
9
  import { reconcileGated, reloadSchedules } from "./cron.mjs";
6
10
  import { makeGitHubAuth } from "./get-token.mjs";
11
+ import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING } from "./processor.mjs";
12
+ import { transientError } from "./transient.mjs";
7
13
  import { makeGitHubHost } from "./github-host.mjs";
8
14
  import { makeGitLabAuth } from "./gitlab-auth.mjs";
9
15
  import { makeGitLabHost } from "./gitlab-host.mjs";
@@ -18,25 +24,34 @@ import { cronFingerprint } from "./fingerprint.mjs";
18
24
  import { makeHostRegistry } from "./host-registry.mjs";
19
25
  import { makeImagePreflight } from "./image-preflight.mjs";
20
26
  import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
27
+ import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, jobUserRefusal, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
21
28
  import { makeCollectChain } from "./outbox.mjs";
22
29
  import { containerPackagePaths, readStageManifest } from "./packages.mjs";
23
30
  import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
24
- import { listRunningSandboxes } from "./sandbox.mjs";
31
+ import { listRunningSandboxes, makeSandboxNetworkSweeper, makeSandboxRuntimeWatch } from "./sandbox.mjs";
32
+ import { makeRetentionSweep } from "./retention-sweep.mjs";
25
33
  import { makeSandboxReaper } from "./sandbox-store.mjs";
26
34
  import { makeSessionStore } from "./session-store.mjs";
35
+ import { scrubCredentials } from "./redact.mjs";
27
36
  import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
37
+ import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "./watch-closer.mjs";
28
38
  import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
29
39
  import { loadScopedLimits, scopeKeyPrefix } from "./scoped-limits.mjs";
40
+ import { makeOnFailure } from "./on-failure.mjs";
30
41
  import { makeWaitChecker } from "./wait-check.mjs";
31
42
  import { makeWaitState } from "./wait-state.mjs";
32
43
  import { hostQueueName, makeQueue } from "./queue.mjs";
33
- import { makeLocalBackend, makeReaper, makeStopContainer } from "./backend-local.mjs";
34
- import { makeBackendRegistry, reapAll } from "./backend-registry.mjs";
35
- import { DEFAULT_BACKEND } from "./backends.mjs";
44
+ import { endpointShown, makeDockerEndpointResolver, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
45
+ import { NETNS_KEEPER_MIN_AGE_MS, NETNS_KEEPER_YOUNG_MARGIN_MS, runtimeFromFacts } from "./netns-keeper.mjs";
46
+ import { makeBackendRegistry, reapAll, resolveBackendName } from "./backend-registry.mjs";
47
+ import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
48
+ import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_INFO_TIMEOUT_MS, cachedPodmanInfo, decidePodmanJobUser, makePodmanBackend, makePodmanInfoReader, makePodmanReaper, observePodman, podmanConfRefusal, unavailableFor, podmanConfWidening, podmanJobUserRefusal, resolvePodmanImageUser } from "./backend-podman.mjs";
49
+ import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootfulMemory, observeHost, observeRootfulConf, readRootfulService, rootfulConfRefusal, rootfulConfRetries, rootfulUnreadList, runtimeObservationKey } from "./runtime-observations.mjs";
36
50
 
37
51
  import { makeRunContainer } from "./run-container.mjs";
52
+ import { resolveProviderCredential } from "./env-allowlist.mjs";
38
53
  import { makeSecretsResolver } from "./secrets.mjs";
39
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
54
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
40
55
  import { makeRunMirror } from "./run-mirror.mjs";
41
56
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
42
57
  import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
@@ -46,6 +61,44 @@ import { makeStallGuard } from "./scheduler-stall-guard.mjs";
46
61
  /** How long boot will wait for `docker image inspect` before shipping without a digest. */
47
62
  const BOOT_IMAGE_TIMEOUT_MS = 5_000;
48
63
 
64
+ /**
65
+ * How long boot waits for the daemon facts read (issues #341 and #345): the image read's 5 s, unless the floor asks for a
66
+ * word only a DAEMON observation earns (`isolation`, `mountSet`), when it waits the facts read's own bound plus 2 s. A
67
+ * busy host's `docker info` is the slow read, and a floor this boot met from the table before must not exit 1 on every
68
+ * restart of a healthy daemon. Exported for its test.
69
+ */
70
+ export function bootFactsBoundMs({ backends, backendFloor }) {
71
+ const needsDaemon = unobservedFloor(backends, backendFloor, { [DOCKER_ENDPOINT_LOCAL]: true }).length > 0;
72
+ return needsDaemon ? DAEMON_FACTS_TIMEOUT_MS + 2_000 : BOOT_IMAGE_TIMEOUT_MS;
73
+ }
74
+
75
+ /**
76
+ * How long after a failed forge-auth re-resolve before another job is allowed to try again (issue #316).
77
+ *
78
+ * The in-flight promise dedupes concurrent callers; this bounds sequential ones. Thirty seconds is short
79
+ * enough that a forge coming back is picked up within one job of it, and long enough that a worker
80
+ * draining a thousand-job backlog against a forge that is still down opens tens of identity calls rather
81
+ * than a thousand -- each of which can otherwise sit on undici's 300-second header timeout with the queue
82
+ * waiting behind it.
83
+ */
84
+ const AUTH_RETRY_COOLDOWN_MS = 30_000;
85
+
86
+ /**
87
+ * How long a job will wait for a forge-auth re-resolve before giving up on it.
88
+ *
89
+ * This is the bound that keeps the re-resolve off the critical path, and without it the feature is a
90
+ * wedge rather than a repair. `mintToken` is awaited inside `runJob`, none of the four identity
91
+ * resolvers passes an `AbortSignal`, and undici's default `headersTimeout` is FIVE MINUTES -- so a forge
92
+ * that accepts the connection and then answers nothing (a load balancer draining, a firewall that drops
93
+ * rather than rejects) would hold every job for that long, with BullMQ renewing the lock the whole time
94
+ * so nothing ever stalls out. At `PI_CONCURRENCY=1` that is the queue stopped, with no error line.
95
+ *
96
+ * Ten seconds is far longer than a healthy `GET /user` and far shorter than anything an operator would
97
+ * call a hang. Losing the race does not cancel the underlying call: it is left to settle into the
98
+ * cooldown, so the next job still benefits from whatever it eventually learns.
99
+ */
100
+ const AUTH_RESOLVE_TIMEOUT_MS = 10_000;
101
+
49
102
  /**
50
103
  * How long a fleet-wide scope claim lives. `JOB_TIMEOUT_MS` plus slack: a container cannot outlive that
51
104
  * ceiling, so the claim cannot expire underneath a live job -- which is what makes a refresh unnecessary
@@ -68,73 +121,123 @@ const WORKER_VERSION = (() => {
68
121
  }
69
122
  })();
70
123
 
124
+ // `makeWatchCloser` lives in its own module since issue #301 (the receiver's triggers watch registers
125
+ // the same handle); re-exported here so every existing importer keeps its address.
126
+ export { makeWatchCloser } from "./watch-closer.mjs";
127
+
71
128
  /**
72
- * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
73
- * the new worker starts draining. A leaked container keeps spending, so it must go before any new
74
- * job launches.
75
- *
76
- * It runs `docker ps` / `docker rm -f` ONLY. It never inspects a container's exit code, never touches
77
- * the queue, and never re-enqueues -- queue and retry state belong to Redis, not to docker
78
- * (INT-RUNNER-EXIT-CODE-PROTOCOL / CONST-RETRY-INFRA-ONLY). It logs container names only (no PII).
129
+ * Race a read against a fuse, and CLEAN THE FUSE UP whichever side wins (issue #300). The fuse is
130
+ * unref'd, deliberately: it exists so a wedged docker daemon cannot hold boot, and it must never itself
131
+ * hold the process. But unref'd is not cleaned up (#295's lesson, both halves): when the read won, the
132
+ * old inline race left its five-second timer armed for the full term. The LOSING read stays pending --
133
+ * a wedged `docker inspect` has no cancel -- which is the read-is-a-nicety posture the call site
134
+ * documents, unchanged here.
79
135
  *
80
- * It assumes ONE worker per docker daemon: a co-located second worker's boot would remove the first's
81
- * in-flight `pi-job-*` container. That is the accepted v1 shape (DES-CONCURRENCY-3, single worker per
82
- * host).
136
+ * NOT FOR BARE CONTEXTS, and the boundary is the fuse's own unref: awaited when nothing else holds the
137
+ * event loop, the fuse never fires and node exits mid-await (measured, exit 13). In this boot the shared
138
+ * redis client and the workers hold the loop, so the fallback always arrives; a caller with an empty
139
+ * loop needs a ref'd timer and a different trade.
83
140
  *
84
- * `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
85
- * `reaper_skipped`, and boot continues to the worker.
141
+ * EXPORTED for the reason `makeWatchCloser` is: the cleared-fuse property is not observable through a
142
+ * full boot without racing every other timer the boot arms, and a guarantee the shutdown story rests on
143
+ * deserves a deterministic pin rather than a census.
86
144
  */
145
+ export async function settleWithin(promise, ms, fallback) {
146
+ let timer = null;
147
+ try {
148
+ return await Promise.race([
149
+ promise,
150
+ new Promise((resolve) => {
151
+ timer = setTimeout(() => resolve(fallback), ms);
152
+ timer.unref?.();
153
+ }),
154
+ ]);
155
+ } finally {
156
+ clearTimeout(timer);
157
+ }
158
+ }
159
+
87
160
  /**
88
161
  * Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
89
162
  * inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
90
- * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
91
- * logs and the worker keeps its boot-time schedulers.
163
+ * `reloadSchedules`. Best-effort: a platform without `fs.watch` logs and the worker keeps its boot-time
164
+ * schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
165
+ * `startWorker` registers so the watch dies with the worker that armed it (issue #295).
92
166
  */
93
- function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
167
+ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot) {
94
168
  const path = config.triggersFile;
95
169
  const dir = dirname(path) || ".";
96
170
  const file = basename(path);
97
- let timer = null;
171
+ const handles = { watcher: null, timer: null, closed: false };
172
+ const closer = makeWatchCloser(handles, log);
173
+ const readFile = () => readFileSync(path, "utf8");
174
+ // The baseline is what the BOOT LOAD read, handed in: the reconcile above and everything between it and
175
+ // this arm -- the endpoint probe, forge auth, the reaper, Valkey -- is the window an edit is lost in
176
+ // (issue #386). A baseline taken here would measure the arming instead.
177
+ readBeforeArming(handles, readFile, atBoot);
98
178
  try {
99
- watch(dir, (_event, changed) => {
179
+ handles.watcher = watch(dir, (_event, changed) => {
180
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
100
181
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
101
- clearTimeout(timer);
102
- timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
103
- }).unref?.();
182
+ clearTimeout(handles.timer);
183
+ handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), WATCH_DEBOUNCE_MS);
184
+ });
185
+ handles.watcher.unref?.();
104
186
  log("triggers_watching", { path });
105
187
  } catch (err) {
106
188
  log("triggers_watch_unavailable", { reason: err?.message });
107
189
  }
190
+ // ONLY WHEN THE BYTES MOVED, because this one costs a Valkey round trip and a reconcile: a quiet boot
191
+ // must not pay for the race it did not lose, and must not log a second `schedules` line saying nothing
192
+ // changed.
193
+ if (changedWhileArming(handles, readFile)) {
194
+ log("triggers_reread_after_arming", { path });
195
+ void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet });
196
+ }
197
+ return closer;
108
198
  }
109
199
 
110
200
  /**
111
201
  * Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
112
202
  * and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
113
- * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort + unref'd.
203
+ * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
204
+ * FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
114
205
  */
115
- function watchPauseWindowsFile(config, ref, log) {
206
+ function watchPauseWindowsFile(config, ref, log, atBoot) {
116
207
  const path = config.pauseWindowsFile;
117
208
  const dir = dirname(path) || ".";
118
209
  const file = basename(path);
119
- let timer = null;
210
+ const handles = { watcher: null, timer: null, closed: false };
211
+ const closer = makeWatchCloser(handles, log);
120
212
  const reload = () => {
121
213
  try {
122
214
  ref.current = loadPauseWindows(config);
123
- log("pause_windows_reloaded", { count: ref.current.length });
215
+ closer.reloadLog("pause_windows_reloaded", { count: ref.current.length });
124
216
  } catch (err) {
125
- log("pause_windows_reload_invalid", { reason: err?.message });
217
+ closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
126
218
  }
127
219
  };
220
+ const readFile = () => readFileSync(path, "utf8");
221
+ readBeforeArming(handles, readFile, atBoot); // the BOOT LOAD's own bytes; see watch-closer (issue #386)
128
222
  try {
129
- watch(dir, (_event, changed) => {
223
+ handles.watcher = watch(dir, (_event, changed) => {
224
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
130
225
  if (changed && changed !== file) return;
131
- clearTimeout(timer);
132
- timer = setTimeout(reload, 150);
133
- }).unref?.();
226
+ clearTimeout(handles.timer);
227
+ handles.timer = setTimeout(reload, WATCH_DEBOUNCE_MS);
228
+ });
229
+ handles.watcher.unref?.();
134
230
  log("pause_windows_watching", { path });
135
231
  } catch (err) {
136
232
  log("pause_windows_watch_unavailable", { reason: err?.message });
137
233
  }
234
+ // SAID, like the triggers watch says it: without a line of its own an operator cannot tell a boot-race
235
+ // reload from an ordinary one, and this is the only reload that happens with nobody editing.
236
+ if (changedWhileArming(handles, readFile)) {
237
+ log("pause_windows_reread_after_arming", { path });
238
+ reload();
239
+ }
240
+ return closer;
138
241
  }
139
242
 
140
243
  /**
@@ -154,23 +257,34 @@ export function reloadScopedLimits(config, ref, log) {
154
257
 
155
258
  /**
156
259
  * Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
157
- * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort + unref'd.
260
+ * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
261
+ * unref'd and the returned closer stops the watch with the worker (issue #295).
158
262
  */
159
- function watchScopedLimitsFile(config, ref, log) {
263
+ function watchScopedLimitsFile(config, ref, log, atBoot) {
160
264
  const path = config.scopedLimitsFile;
161
265
  const dir = dirname(path) || ".";
162
266
  const file = basename(path);
163
- let timer = null;
267
+ const handles = { watcher: null, timer: null, closed: false };
268
+ const closer = makeWatchCloser(handles, log);
269
+ const readFile = () => readFileSync(path, "utf8");
270
+ readBeforeArming(handles, readFile, atBoot); // the BOOT LOAD's own bytes; see watch-closer (issue #386)
164
271
  try {
165
- watch(dir, (_event, changed) => {
272
+ handles.watcher = watch(dir, (_event, changed) => {
273
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
166
274
  if (changed && changed !== file) return;
167
- clearTimeout(timer);
168
- timer = setTimeout(() => reloadScopedLimits(config, ref, log), 150);
169
- }).unref?.();
275
+ clearTimeout(handles.timer);
276
+ handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
277
+ });
278
+ handles.watcher.unref?.();
170
279
  log("scoped_limits_watching", { path });
171
280
  } catch (err) {
172
281
  log("scoped_limits_watch_unavailable", { reason: err?.message });
173
282
  }
283
+ if (changedWhileArming(handles, readFile)) {
284
+ log("scoped_limits_reread_after_arming", { path });
285
+ reloadScopedLimits(config, ref, closer.reloadLog);
286
+ }
287
+ return closer;
174
288
  }
175
289
 
176
290
  /**
@@ -200,34 +314,118 @@ export async function startWorker(
200
314
  // tests in `start-wiring.test.mjs` were reported as not existing at all -- no name, no count, exit 0 --
201
315
  // because this function's log line went through the same channel the runner needed (issue #266).
202
316
  write = (chunk) => process.stdout.write(chunk),
317
+ // Issue #354: the configuration loader, a seam for ONE reason: a test venue that is not in the backend table (an
318
+ // injected `extraBackends` bundle) cannot be written through the real loader, which refuses a name it does not know.
319
+ // The table's own venues (`local`, and `podman` since part 2) go through the real loader.
320
+ loadConfig: loadConfigFn = loadConfig,
203
321
  makeAuth = makeGitHubAuth,
204
322
  makeHost = makeGitHubHost,
205
323
  createWorkerFn = createWorker,
206
324
  makeReaper: makeReaperFn = makeReaper,
207
325
  makeBackendRegistry: makeBackendRegistryFn = makeBackendRegistry,
208
- // Additional backend bundles, in registration order after `local`. The deployment still decides which
209
- // are BLESSED (PI_BACKENDS) and which is default; this only says which exist.
326
+ // Additional backend bundles, in registration order after `local` (which is built only while blessed,
327
+ // issue #354). The deployment still decides which are BLESSED (PI_BACKENDS) and which is default; this only
328
+ // says which exist.
210
329
  extraBackends = [],
211
330
  makeLogSink: makeLogSinkFn = makeLogSink,
212
331
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
213
332
  makeRunMirror: makeRunMirrorFn = makeRunMirror,
214
333
  makeLogReaper: makeLogReaperFn = makeLogReaper,
215
334
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
335
+ makeSandboxNetworkSweeper: makeSandboxNetworkSweeperFn = makeSandboxNetworkSweeper,
336
+ // Issue #429: the one call that asks a runtime which sandboxes are open, seamed so a wiring test can drive the real
337
+ // sandbox reaper against a real retention root without spawning docker or podman.
338
+ listRunningSandboxes: listRunningSandboxesFn = listRunningSandboxes,
339
+ makeRetentionSweep: makeRetentionSweepFn = makeRetentionSweep,
216
340
  makeRunContainer: makeRunContainerFn = makeRunContainer,
217
341
  makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
218
342
  makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
219
343
  makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
220
344
  makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
221
345
  makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
346
+ // Which docker endpoint this host's CLI resolves (issue #278). A seam because the real one spawns the
347
+ // docker CLI, and a wiring test must decide what it answers.
348
+ resolveDockerEndpoint: resolveDockerEndpointFn = makeDockerEndpointResolver(),
349
+ // Issue #341: the one `docker info` the job-user decision reads, and the process facts it reads beside it.
350
+ // Seams for the same reason as the endpoint: a wiring test decides what the daemon and the process say.
351
+ readDaemonFacts: readDaemonFactsFn = makeDaemonFactsReader(),
352
+ // `home` (issue #354) is the account whose rootless Podman runs the podman venue's jobs: its own mounts.conf and
353
+ // containers.conf are read from there. Absent in a test's identity, it falls to the observation's own default.
354
+ jobUserIdentity = { platform: process.platform, release: osRelease(), euid: process.geteuid?.(), egid: process.getegid?.(), home: homedir() },
355
+ // Issue #345: the host files the runtime-mounts observation reads (Podman's mounts.conf and containers.conf). A seam so
356
+ // a wiring test never reads this machine's /etc.
357
+ observationFs = { statSync, readFileSync, readdirSync },
358
+ // Issue #448: `systemctl show podman.service`, read only where a local job runs on rootful Podman's Docker API on this
359
+ // host. A seam so a wiring test decides what the unit says, and never asks this machine's systemd.
360
+ readPodmanService: readPodmanServiceFn = makePodmanServiceReader(),
361
+ // Issue #354: the podman venue's three boot collaborators, each a seam for the endpoint's reason: the real ones spawn
362
+ // `podman`, and a wiring test must decide what `podman info` says, what the reaper lists and what the bundle is
363
+ // built from. Constructing the default reader spawns nothing; only a call does, and only while `podman` is blessed.
364
+ readPodmanInfo: readPodmanInfoFn = makePodmanInfoReader(),
365
+ makePodmanReaper: makePodmanReaperFn = makePodmanReaper,
366
+ makePodmanBackend: makePodmanBackendFn = makePodmanBackend,
222
367
  makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
223
368
  makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
224
369
  makeForgejoAuth: makeForgejoAuthFn = makeForgejoAuth,
225
370
  makeForgejoHost: makeForgejoHostFn = makeForgejoHost,
226
371
  makeAzureAuth: makeAzureAuthFn = makeAzureAuth,
227
372
  makeAzureHost: makeAzureHostFn = makeAzureHost,
373
+ // The clock the forge-auth re-resolve cooldown reads. Injected because a test that asserts a window
374
+ // beside a subject built on the default `Date.now` is a fuse: it passes until the wall clock drifts
375
+ // past that window, then fails in CI on a tree nobody touched (issue #284).
376
+ now = () => Date.now(),
377
+ // Issue #476: how the boot waits out a rootless network keeper that is only too young. A seam so a wiring test
378
+ // proves the wait and its bound without spending real seconds.
379
+ sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms)),
380
+ // How long a job waits for a forge-auth re-resolve. A seam rather than a constant because the
381
+ // property under test is that the bound EXISTS, and a test that proved it by waiting ten real
382
+ // seconds would be paid for on every run for the life of the file.
383
+ authResolveTimeoutMs = AUTH_RESOLVE_TIMEOUT_MS,
384
+ // Issue #464: the jobs dir, created and checked as this account's at boot. A seam so a wiring test decides it.
385
+ ensureJobsDir: ensureJobsDirFn = ensureJobsDir,
386
+ // Issue #464: the same rule for the two durable stores, which default into that root on an account with no home.
387
+ ensureUnderAccountRoot: ensureUnderAccountRootFn = ensureUnderAccountRoot,
388
+ // Issue #464 (gate round 1): the sandbox dir, refused at boot when another account owns it, as the jobs dir is.
389
+ ensureSandboxDir: ensureSandboxDirFn = ensureSandboxDir,
390
+ // Issue #464 (gate round 2): whose Valkey VALKEY_URL reaches, judged here, where the connection is made, and the
391
+ // literal address every Valkey client of this worker then connects to. A seam: the real one reads /proc, probes
392
+ // this host's addresses and resolves the name, none of which belongs in a wiring test.
393
+ judgeValkey = defaultJudgeValkey,
228
394
  } = {},
229
395
  ) {
230
- const config = loadConfig(env);
396
+ const config = loadConfigFn(env);
397
+ // Issue #354: whether this deployment blesses the local adapter (its name is `DEFAULT_BACKEND`, the word
398
+ // `makeLocalBackend` stamps on its bundle). `local` stopped being mandatory in `PI_BACKENDS`, and every read
399
+ // below that asks THIS HOST'S DOCKER CLI something is a read about that one venue: its endpoint, its daemon's
400
+ // facts, its image, its reaper, its sandboxes. With `local` unblessed none of them runs, so a host without
401
+ // Docker never spawns `docker` at boot for a venue it will never use, and a floor naming an observation only
402
+ // `local` makes cannot hold such a worker at exit 1 forever. Everything is byte-identical while it is blessed.
403
+ const localBlessed = config.backends.includes(DEFAULT_BACKEND);
404
+ // The same split for the native podman venue (issue #354, `DES-PODMAN-NATIVE-ROOTLESS-BACKEND`): its `podman info`
405
+ // read, its reaper and its bundle exist only while it is blessed, so a deployment that never chose it spawns no
406
+ // `podman` and pays nothing for it being in the table.
407
+ const podmanBlessed = config.backends.includes(PODMAN_BACKEND);
408
+ // Issue #354: an ABSENT `observationPreflight` admits every job, which is the right answer only for a venue whose
409
+ // words hold without observing anything. A venue whose table entry names an `observedBy` and carries no preflight
410
+ // would have those words hold with nothing looking, and a floor naming one would pass on capability alone: the
411
+ // believed-in control `unarmedFloor` was written against. The registry cannot read the table, so it is refused
412
+ // here, and HERE rather than beside the registry: this runs before anything is spawned or connected, so the refusal
413
+ // leaves nothing open behind it. Only the extra bundles need it; `local`'s is built below carrying its preflight.
414
+ for (const bundle of extraBackends) {
415
+ // A string name only: `backendFor(undefined)` answers with the table's default, and a nameless bundle is the
416
+ // registry's refusal to make, with its own message.
417
+ const gated = typeof bundle?.name === "string" ? Object.keys(backendFor(bundle.name)?.observedBy ?? {}) : [];
418
+ if (gated.length > 0 && typeof bundle?.observationPreflight !== "function") {
419
+ throw new Error(`backend ${JSON.stringify(bundle.name)} declares ${gated.join(", ")} as held only while observed, and carries no observationPreflight to observe them`);
420
+ }
421
+ }
422
+ // Issue #464: the jobs dir is this account's, or the worker does not start. Before it built anything, and a
423
+ // `configError` (exit 2, never restarted) where another account owns it, since a restart meets the same owner: the
424
+ // shared `<tmp>/pi-dispatch/jobs` default let one account's jobs dir fail every other account's jobs with EACCES
425
+ // while the worker ran on. A failed mkdir is the fs error it is (exit 1).
426
+ ensureJobsDirFn(config.jobsDir, { env });
427
+ ensureSandboxDirFn(config.sandboxDir, { env });
428
+ for (const store of [config.logsDir, dirname(config.settingsFile)]) ensureUnderAccountRootFn(store, { env });
231
429
  // `host` sits AFTER the spread, so it is authoritative rather than overridable (issue #57). No call
232
430
  // site can know better than this closure which process wrote a line, and one that passed `host` would
233
431
  // be lying by construction -- verified: none does. This is also why the stamp lives ONLY here. Every
@@ -236,6 +434,22 @@ export async function startWorker(
236
434
  // while one added inside this closure cannot reach them.
237
435
  const log = (event, fields = {}) => write(`${JSON.stringify({ event, ...fields, host: config.workerName })}\n`);
238
436
 
437
+ // Issue #464 (gate round 2): the owner rule where the connection is made. `service install`, `up` and doctor judge
438
+ // the Valkey too, but a VALKEY_URL edited after they refused it, a 0.0.0.0 URL, or another account publishing on ::1
439
+ // after the install each reached another account's queue through an unjudged worker (measured on Fedora 44). So the
440
+ // worker judges it itself, before it contacts Valkey at all, with the same function (`resolveWorkerValkey`), and
441
+ // every client below connects to the judged literal address, never to a name resolved again later. A refusal is a
442
+ // configError (exit 2, not restarted into the same answer); nothing answering is a plain error (exit 1, restarted).
443
+ const valkey = await judgeValkey({ url: config.valkeyUrl, venues: { localUsed: localBlessed, podmanUsed: podmanBlessed }, env });
444
+ for (const note of valkey.notes ?? []) log("valkey_note", { note });
445
+ if (valkey.pinned) log("valkey_pinned", { address: valkey.pinned.address, port: valkey.pinned.port, heldBy: valkey.pinned.heldBy });
446
+ // Every client below connects through connection.mjs' JudgedConnector, which judges the (pinned) address again on
447
+ // each connect, as this boot judged it: root refused as the boot decided it, PI_VALKEY_SHARED from the deployment .env.
448
+ const valkeyContext = workerValkeyContext(valkey, env);
449
+ // Issue #468: whether this worker sends a password, and never the password: the only thing a log line may say of it.
450
+ log("valkey_password", { set: Boolean(valkeyPasswordFor(valkey.url, valkeyContext).password) });
451
+ const valkeyConn = (opts = {}) => parseConnection(valkey.url, { ...opts, servername: valkey.servername, context: valkeyContext });
452
+
239
453
  // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
240
454
  // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
241
455
  // than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
@@ -244,16 +458,172 @@ export async function startWorker(
244
458
  // scheduled, and a `const` frozen at boot would make it publish the pre-edit set forever -- so two
245
459
  // hosts would see each other's fingerprint oscillate on the beat period, refusing or agreeing
246
460
  // depending on which half of a beat a reload happened to land in.
247
- const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared }) };
461
+ // THE BOOT LOADS' OWN READS are the baselines for the three watches' boot-race checks (issue #386),
462
+ // captured through the seam each loader already has rather than re-read where the watch arms. The window
463
+ // this race lives in is between these lines and the arming a thousand lines below -- the endpoint probe,
464
+ // forge auth, the reaper, Valkey -- not the microseconds around the arming itself, which is what a first
465
+ // attempt measured. `null` where a file is not configured, which reads as "nothing to compare".
466
+ const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null };
467
+ const recording = (into, path) => ({
468
+ readFileSync: (file, enc) => {
469
+ const text = readFileSync(file, enc);
470
+ if (file === path) atBoot[into] = text;
471
+ return text;
472
+ },
473
+ });
474
+ const schedules = { current: loadSchedules(config, { fleet: config.workerNameDeclared, ...recording("triggers", config.triggersFile) }) };
248
475
 
249
476
  // REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
250
477
  // Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
251
478
  // pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
252
- const pauseWindows = { current: loadPauseWindows(config) };
479
+ const pauseWindows = { current: loadPauseWindows(config, recording("pauseWindows", config.pauseWindowsFile)) };
253
480
 
254
481
  // Issue #242: same posture for the scoped-limits file -- fail-loud with the operator present, mutable
255
482
  // ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
256
- const scopedLimits = { current: loadScopedLimits(config) };
483
+ const scopedLimits = { current: loadScopedLimits(config, recording("scopedLimits", config.scopedLimitsFile)) };
484
+
485
+ // Issue #278: WHICH DOCKER DAEMON the job containers' credentials will travel to. Asked of the CLI at boot,
486
+ // AFTER the free file validations above (a wedged CLI costs up to its bound, and must not delay them) and
487
+ // BEFORE forge auth, the reaper, Valkey and the worker -- so a refusal here stops a process that built
488
+ // nothing. Without a floor naming credentialTransit a redirect is words only: pointing the worker at a
489
+ // daemon is the operator's call. With one asking for `enforced`, it refuses: a floor is not met by capability
490
+ // alone. A TRANSIENT failure to ask (a timeout, a spawn out of resources) throws untagged, exit 1, so the
491
+ // supervisor retries; everything else is a config error, exit 2.
492
+ //
493
+ // Issue #354: judged for `local` ALONE, not for every blessed venue. These are observations of this host's docker
494
+ // CLI and its daemon, which mean nothing for another venue's words: a venue with its own `observedBy` brings its own
495
+ // boot read (the podman venue's is below, judged for `podman` alone the same way).
496
+ const bootEndpoint = localBlessed ? await resolveDockerEndpointFn() : null;
497
+ if (bootEndpoint) {
498
+ logDockerEndpoint(log, bootEndpoint);
499
+ const [endpointRefusal] = observationRefusals({
500
+ backends: [DEFAULT_BACKEND],
501
+ backendFloor: config.backendFloor,
502
+ observations: { [DOCKER_ENDPOINT_LOCAL]: bootEndpoint.local === true },
503
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(bootEndpoint) },
504
+ // The daemon has not been read yet; the runtime observations are checked once it has (issue #345).
505
+ only: [DOCKER_ENDPOINT_LOCAL],
506
+ });
507
+ if (endpointRefusal) throw bootEndpoint.local === null && bootEndpoint.transient ? new Error(endpointRefusal) : configError(endpointRefusal);
508
+ }
509
+ // The per-job read (below, `observationPreflight`) logs only when the answer CHANGES from the last one, so a
510
+ // deliberate, standing redirect writes one line at boot rather than one per job.
511
+ let endpointSeen = bootEndpoint ? dockerEndpointState(bootEndpoint) : null;
512
+
513
+ // Issue #341: WHO job containers run as on this daemon (`DES-JOB-USER-INFERRED-READ-BACK-ON-REQUEST`). Decided
514
+ // from facts, never a probe container, cached per endpoint state. Bounded like the boot image read, because a
515
+ // wedged daemon must not hang boot. Only an IDENTITY verdict refuses at boot, and only when `local` is the
516
+ // default venue: rootless, userns-remap, a root worker and Docker Desktop on Linux cannot run any local job here.
517
+ // An unknown answer (a daemon still starting) boots, so a unit with RestartPreventExitStatus=2 is never stranded
518
+ // by one; so does `runtime-unreadable`, which a later job re-reads.
519
+ const resolveJobUser = makeJobUserResolver({ readFacts: readDaemonFactsFn, ...jobUserIdentity });
520
+ // Issue #345: a floor that needs a DAEMON observation waits the facts read's own bound (plus a margin), not the image
521
+ // read's 5 s: a busy host's `docker info` is the slow read, and a floor this boot met from the table before would
522
+ // otherwise exit 1 on every restart of a healthy daemon.
523
+ // Issue #354: `null` throughout with `local` unblessed. There is no local job to decide a user for, so nothing is
524
+ // read, nothing is said, and the boot line carries `jobUser: null` rather than a decision about a daemon nobody asked.
525
+ const bootJobUser = bootEndpoint
526
+ ? await settleWithin(
527
+ resolveJobUser({ endpoint: bootEndpoint, key: endpointSeen }).catch(() => null),
528
+ bootFactsBoundMs(config),
529
+ null,
530
+ )
531
+ : null;
532
+ const bootDecision = bootEndpoint ? (bootJobUser?.decision ?? { mode: "unknown", user: null, cause: null, reason: "boot-read-timeout" }) : null;
533
+ if (bootDecision) log("job_user", { mode: bootDecision.mode, user: bootDecision.user, cause: bootDecision.cause, reason: bootDecision.reason });
534
+ // Said again whenever a job's decision differs from the last one said, so a boot that read `unknown` (a daemon still
535
+ // starting) and a later answer that refuses every job are never separated by silence.
536
+ let jobUserSaid = bootDecision ? jobUserLogKey(bootDecision) : null;
537
+
538
+ // Issue #345: the RUNTIME observations, from that same facts read and this host's files, checked against the floor
539
+ // the way the endpoint was above: `isolation` holds only while the daemon is observed applying a container's bounds,
540
+ // `mountSet` only while the runtime is observed adding no mounts of its own. Without a floor naming either, this is
541
+ // words (worker_started, and doctor). A refusal resting only on a read that did not answer (a daemon still starting,
542
+ // the boot bound) throws untagged, exit 1, so the supervisor retries; one resting on an answer is a config error.
543
+ // Issue #354: for `local` alone, like the endpoint above, and not at all with `local` unblessed. This is the site that
544
+ // would otherwise hold a worker without `local` at exit 1 forever: a floor naming `isolation` against observations
545
+ // nobody made reads as unanswered, which is the transient arm, which the supervisor retries without end.
546
+ // Issue #448: `systemctl show podman.service`, read once here where the local daemon is rootful Podman on this host, and
547
+ // handed to both the mounts observation and the containers.conf refusal below, so they judge one answer.
548
+ // The deletion rule's memory (gate round 1 of PR #473): which chain files this worker saw while one service start ran.
549
+ const rootfulMemory = makeRootfulMemory();
550
+ const bootUnit = bootEndpoint && bootJobUser?.daemon ? await readRootfulService({ endpoint: bootEndpoint, daemon: bootJobUser.daemon, readService: readPodmanServiceFn }) : undefined;
551
+ const bootObserved = bootEndpoint ? observeHost({ endpoint: bootEndpoint, daemon: bootJobUser?.daemon ?? { answered: false, reason: "boot-read-timeout", transient: true }, fs: observationFs, unit: bootUnit, env, memory: rootfulMemory }) : null;
552
+ if (bootObserved) {
553
+ const bootObservedArgs = { backends: [DEFAULT_BACKEND], backendFloor: config.backendFloor, observations: bootObserved.observations, evidence: bootObserved.evidence };
554
+ const [runtimeRefusal] = observationRefusals(bootObservedArgs);
555
+ if (runtimeRefusal) throw observationRefusalIsTransient(bootObservedArgs) ? new Error(runtimeRefusal) : configError(runtimeRefusal);
556
+ }
557
+ let runtimeObservedSaid = bootObserved ? runtimeObservationKey(bootObserved) : null;
558
+
559
+ const bootRefusal = jobUserBootRefusal(bootDecision, config.defaultBackend);
560
+ if (bootRefusal) throw configError(bootRefusal);
561
+ // Issue #448: where a local job runs on rootful Podman's Docker API service on this host, the containers.conf that
562
+ // service reads (the system chain, root's own, the unit's CONTAINERS_CONF) and whether the running service started
563
+ // before it last changed. A VENUE refusal, as the podman venue's #428 one is and for its reason: with egress off no
564
+ // declared property covers what `env` or `annotations` add to a job. After the identity, whose fix comes first, and
565
+ // only while `local` is the default venue; merely blessed, each local job is refused and the default venue's run.
566
+ // Tagged (exit 2) since a restart reads the same bytes, except what heals by itself (exit 1): a read that failed for a
567
+ // moment, a running service older than its containers.conf, a change time ahead of the clock. What the worker's account
568
+ // cannot read under root's own config home is said once here, never refused on; any other unreadable part refuses.
569
+ const bootRootful = bootEndpoint && bootJobUser?.daemon ? await observeRootfulConf({ endpoint: bootEndpoint, daemon: bootJobUser.daemon, fs: observationFs, readService: readPodmanServiceFn, env, unit: bootUnit, memory: rootfulMemory }) : null;
570
+ let rootfulUnreadSaid = "";
571
+ const sayRootfulUnread = (rootful) => {
572
+ const said = rootfulUnreadList(rootful?.unread);
573
+ if (said === rootfulUnreadSaid) return;
574
+ rootfulUnreadSaid = said;
575
+ if (said !== "") log("local_podman_conf_unread", { unread: said });
576
+ };
577
+ sayRootfulUnread(bootRootful);
578
+ const localConfBoot = localConfBootRefusal(bootDecision, config.defaultBackend, bootRootful);
579
+ if (localConfBoot) throw localConfBoot.transient ? new Error(localConfBoot.message) : configError(localConfBoot.message);
580
+
581
+ // Issue #354: the podman venue's ONE facts read, `podman info --format json`, at boot, where local's reads are and for
582
+ // their reasons: free, before forge auth, the reaper and any Valkey client (the Valkey owner judgement above only reads
583
+ // sockets and probes addresses), so a refusal here stops a process that built nothing.
584
+ // Wrapped once in `cachedPodmanInfo` and the SAME wrapper is handed to the bundle below, so an answer read here is the
585
+ // one the first job is decided from rather than a second spawn. Bounded twice: the reader's own timeout (docker info's
586
+ // 15 s, reused) and this fuse two seconds past it, because a spawn that never settles has no timeout to fire.
587
+ const podmanInfo = podmanBlessed ? cachedPodmanInfo(readPodmanInfoFn) : null;
588
+ const bootPodmanRead = podmanInfo
589
+ ? await settleWithin(
590
+ Promise.resolve()
591
+ .then(() => podmanInfo())
592
+ .catch(() => ({ answered: false, reason: "spawn-failed", transient: true })),
593
+ PODMAN_INFO_TIMEOUT_MS + 2_000,
594
+ { answered: false, reason: "boot-read-timeout", transient: true },
595
+ )
596
+ : null;
597
+ const bootPodmanDecision = bootPodmanRead ? decidePodmanJobUser({ platform: jobUserIdentity.platform, euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, read: bootPodmanRead }) : null;
598
+ // The IDENTITY verdict first, before the observations, and unlike `local`'s order on purpose: a rootful or remote
599
+ // Podman also fails the observations (the mounts check reads a rootless user's files, the service one its remoteness),
600
+ // and a floor refusal naming `podmanAddsNoMounts` would send the operator after a mounts.conf when the fix is the
601
+ // account Podman runs as. Only when `podman` is the DEFAULT venue, as `jobUserBootRefusal` for `local`: with it merely
602
+ // blessed, the jobs that name it are refused one by one and the default venue's still run.
603
+ const podmanRefusal = podmanBootRefusal(bootPodmanDecision, config.defaultBackend);
604
+ if (podmanRefusal) throw configError(podmanRefusal);
605
+ // Issue #428: the account's containers.conf, next, under the same rule (only while `podman` is the default venue; merely
606
+ // blessed, each podman job is refused and the default venue's still run). A VENUE refusal rather than a floor
607
+ // observation: with egress off no declared property covers what a job reaches on the host, so a floor would never
608
+ // fire on the deployments at risk. Read from this host's files with no daemon, so it is determinate whatever the info
609
+ // read said, and tagged (exit 2): a restart reads the same bytes. After the identity, whose fix comes first. The one
610
+ // exception is a read that failed for a moment (out of descriptors, an I/O error), which is untagged (exit 1) so the
611
+ // supervisor restarts it, as an unanswered observation is.
612
+ const podmanConfBoot = podmanConfBootRefusal(bootPodmanDecision, config.defaultBackend, { fs: observationFs, home: jobUserIdentity.home, env, euid: jobUserIdentity.euid, runRoot: bootPodmanRead?.answered === true && bootPodmanRead.info ? (bootPodmanRead.info.runRoot ?? null) : undefined });
613
+ if (podmanConfBoot) throw podmanConfBoot.transient ? new Error(podmanConfBoot.message) : configError(podmanConfBoot.message);
614
+ // The podman venue's own observations, judged for `podman` ALONE, as the endpoint and the daemon's are for `local`
615
+ // alone: each venue's words are earned by its own reads, and judging one venue's answers over every blessed venue
616
+ // would read the other's observations as unanswered, which is the transient arm, exit 1 on every restart. Same split
617
+ // as `local`'s: a refusal resting only on a read that did not answer is untagged (retried), one on an answer tagged.
618
+ const bootPodmanObserved = bootPodmanRead ? observePodman({ read: bootPodmanRead, fs: observationFs, home: jobUserIdentity.home, env, euid: jobUserIdentity.euid }) : null;
619
+ // Not judged at all when the venue's identity is already refused (it is then merely blessed, or the refusal above would
620
+ // have stopped the boot): every job naming it is refused by that cause, and a floor refusal here would stop the whole
621
+ // worker, the default venue's jobs included, with a fix (delegate controllers, empty a mounts.conf) that is not the one.
622
+ if (bootPodmanObserved && bootPodmanDecision?.mode !== "unmappable") {
623
+ const podmanObservedArgs = { backends: [PODMAN_BACKEND], backendFloor: config.backendFloor, observations: bootPodmanObserved.observations, evidence: bootPodmanObserved.evidence };
624
+ const [podmanObservationRefusal] = observationRefusals(podmanObservedArgs);
625
+ if (podmanObservationRefusal) throw observationRefusalIsTransient(podmanObservedArgs) ? new Error(podmanObservationRefusal) : configError(podmanObservationRefusal);
626
+ }
257
627
 
258
628
  // The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
259
629
  // Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
@@ -264,24 +634,137 @@ export async function startWorker(
264
634
  // Auth stays BEST-EFFORT per forge, exactly as it was: a local-only deployment has no GitHub
265
635
  // credentials and must still boot and drain cron jobs. The refusal is deferred to the job that needs
266
636
  // the missing credential (the mintToken fallback below), not raised at startup.
637
+ //
638
+ // WHAT CHANGED (issue #316): best-effort used to mean best-effort ONCE. Any throw left `auth` null for
639
+ // the lifetime of the process, so a forge that was merely unreachable during the seconds this loop ran
640
+ // stayed credential-less until somebody restarted the worker, and every job of that kind then hit the
641
+ // `configError` fallback below -- which, since #310, refunds the reserve and posts a public comment
642
+ // telling the issue author that the operator's deployment is misconfigured. A deployment that
643
+ // `doctor` reports as healthy. The boot posture is unchanged for a DETERMINATE failure, which is what
644
+ // the local-only case is (no `gh` on PATH is `ENOENT`); a TRANSIENT one now leaves a re-resolver
645
+ // behind instead of a permanent null.
267
646
  const forges = { github: { auth: null, host: makeHost() } };
268
- try {
269
- forges.github.auth = await makeAuth(config.github);
270
- log("self_identity", { kind: "github", id: forges.github.auth.selfId, source: forges.github.auth.source });
271
- } catch (err) {
272
- log("github_auth_unavailable", { kind: "github", reason: err?.message });
273
- }
647
+ // Per forge kind, what it would take to resolve its auth again: the closure, and the `idOf` its log
648
+ // line needs. Present only while the last attempt failed transiently -- a determinate failure removes
649
+ // it, because retrying a wrong credential is how a deployment pays to be told the same thing twice.
650
+ const authRetries = new Map();
651
+ const authInFlight = new Map();
652
+ // When the last transient attempt happened, so a backlog draining against a forge that is still down
653
+ // does not open one identity round-trip per job. The in-flight promise below dedupes CONCURRENT
654
+ // callers; this bounds SEQUENTIAL ones, which is the shape a PI_CONCURRENCY=1 worker actually has.
655
+ const authCooldownUntil = new Map();
656
+ const authLastError = new Map();
657
+
658
+ /**
659
+ * Boot-time attempt for one forge, and the record of what to do if it failed.
660
+ *
661
+ * `idOf` exists because azure's selfId is an object while the other three are scalars, which is the
662
+ * one place a forge identity does not reduce to a single value.
663
+ */
664
+ const attachAuth = async (kind, make, cfg, idOf = (auth) => auth.selfId) => {
665
+ try {
666
+ forges[kind].auth = await make(cfg);
667
+ log("self_identity", { kind, id: idOf(forges[kind].auth), source: forges[kind].auth.source });
668
+ } catch (err) {
669
+ // The tag is the whole discriminator, and it is now trustworthy at these sites: the identity
670
+ // modules throw untagged for a fetch rejection, a transient status and an unparseable body.
671
+ const transient = err?.piDispatchConfig !== true;
672
+ if (transient) authRetries.set(kind, { resolve: () => make(cfg), idOf });
673
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient });
674
+ }
675
+ };
676
+
677
+ /**
678
+ * The auth for a forge, resolving it now if boot could not and the reason was transient.
679
+ *
680
+ * THREE bounds, because "ask again" is a live network round-trip on the job path and each of them is a
681
+ * different way of asking too often.
682
+ *
683
+ * ONE in-flight promise per forge dedupes CONCURRENT callers, which is what a worker at
684
+ * PI_CONCURRENCY>1 draining a backlog produces. A COOLDOWN bounds sequential ones: without it a
685
+ * PI_CONCURRENCY=1 worker chewing through a backlog against a forge that is still down opens one
686
+ * identity call per job, forever, and each of them can sit on undici's 300-second header timeout with
687
+ * the queue behind it. Inside the cooldown the last error is re-thrown immediately, which is the same
688
+ * verdict at none of the cost. And a DETERMINATE answer retires the re-resolver entirely.
689
+ *
690
+ * The determinate error is RETHROWN rather than turned into `null`. Returning null sent every such job
691
+ * to the generic "configure GITHUB_AUTH_SOURCE" message while the specific reason -- bad credentials,
692
+ * a key that is not PKCS//8 -- was already in hand and went only to the log. The caller decides what to
693
+ * do with it; the point is that it reaches the caller.
694
+ */
695
+ const ensureAuth = async (kind) => {
696
+ if (forges[kind]?.auth) return forges[kind].auth;
697
+ const retry = authRetries.get(kind);
698
+ if (!retry) return null;
699
+ if (!authInFlight.has(kind)) {
700
+ const until = authCooldownUntil.get(kind) ?? 0;
701
+ if (now() < until) throw authLastError.get(kind) ?? transientError(`${kind} auth is still unavailable`);
702
+ const attempt = async () => {
703
+ try {
704
+ const auth = await retry.resolve();
705
+ forges[kind].auth = auth;
706
+ authCooldownUntil.delete(kind);
707
+ authLastError.delete(kind);
708
+ log("self_identity", { kind, id: retry.idOf(auth), source: auth.source });
709
+ return auth;
710
+ } catch (err) {
711
+ authLastError.set(kind, err);
712
+ if (err?.piDispatchConfig === true) {
713
+ authRetries.delete(kind);
714
+ log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient: false });
715
+ } else {
716
+ authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
717
+ }
718
+ throw err;
719
+ }
720
+ };
721
+ // The cleanup is chained OUTSIDE the async body rather than written as its `finally`, and that is
722
+ // not a style choice. An async function runs synchronously up to its first `await`, so a factory
723
+ // that throws SYNCHRONOUSLY runs the whole body -- catch and finally included -- before this
724
+ // `set` ever happens: the delete would find an empty map and the rejected promise would then be
725
+ // installed permanently, leaving that forge dead for the lifetime of the process. Which is the
726
+ // exact defect issue #316 exists to remove, reintroduced by its own fix. A `.finally` callback
727
+ // is always a microtask, so it cannot outrun the `set`, and the identity check makes it safe
728
+ // against a later attempt having already replaced the entry.
729
+ const inflight = attempt().finally(() => {
730
+ if (authInFlight.get(kind) === inflight) authInFlight.delete(kind);
731
+ });
732
+ // A handler, so that a caller losing the timeout race below cannot turn this into an unhandled
733
+ // rejection. Every awaiter still sees the rejection through its own `await`.
734
+ inflight.catch(() => {});
735
+ authInFlight.set(kind, inflight);
736
+ }
737
+ const inflight = authInFlight.get(kind);
738
+ let timer;
739
+ try {
740
+ return await Promise.race([
741
+ inflight,
742
+ new Promise((_resolve, reject) => {
743
+ timer = setTimeout(() => reject(transientError(`${kind} auth did not answer within ${authResolveTimeoutMs}ms`)), authResolveTimeoutMs);
744
+ timer.unref?.();
745
+ }),
746
+ ]);
747
+ } catch (err) {
748
+ // A lost race is a forge that is not answering, which is exactly what the cooldown is for: the
749
+ // in-flight call is still out there and will set it when it settles, but the next job must not
750
+ // queue up behind it in the meantime.
751
+ if (!authLastError.has(kind)) {
752
+ authLastError.set(kind, err);
753
+ authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
754
+ }
755
+ throw err;
756
+ } finally {
757
+ clearTimeout(timer);
758
+ }
759
+ };
760
+
761
+ await attachAuth("github", makeAuth, config.github);
274
762
  // GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
275
763
  // with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
276
764
  // answer than an entry that exists and cannot authenticate.
277
765
  if (config.gitlab) {
278
766
  forges.gitlab = { auth: null, host: makeGitLabHostFn({ apiUrl: config.gitlab.apiUrl }) };
279
- try {
280
- forges.gitlab.auth = await makeGitLabAuthFn(config.gitlab);
281
- log("self_identity", { kind: "gitlab", id: forges.gitlab.auth.selfId, source: forges.gitlab.auth.source });
282
- } catch (err) {
283
- log("gitlab_auth_unavailable", { kind: "gitlab", reason: err?.message });
284
- }
767
+ await attachAuth("gitlab", makeGitLabAuthFn, config.gitlab);
285
768
  }
286
769
  // Forgejo joins on the same best-effort terms. Its auth can fail for one reason the others cannot: a
287
770
  // repository-scoped token cannot call GET /user, so an operator who scoped their token without setting
@@ -289,24 +772,14 @@ export async function startWorker(
289
772
  // credential-less, which refuses its jobs at mint time rather than running them unattributed.
290
773
  if (config.forgejo) {
291
774
  forges.forgejo = { auth: null, host: makeForgejoHostFn({ apiUrl: config.forgejo.apiUrl }) };
292
- try {
293
- forges.forgejo.auth = await makeForgejoAuthFn(config.forgejo);
294
- log("self_identity", { kind: "forgejo", id: forges.forgejo.auth.selfId, source: forges.forgejo.auth.source });
295
- } catch (err) {
296
- log("forgejo_auth_unavailable", { kind: "forgejo", reason: err?.message });
297
- }
775
+ await attachAuth("forgejo", makeForgejoAuthFn, config.forgejo);
298
776
  }
299
777
  // Azure joins on the same terms. Its selfId is an OBJECT (`{ id, email }`) rather than a scalar, because
300
778
  // a pull-request delivery names an actor by GUID and a work item names them only by address -- the one
301
779
  // place a forge's identity does not reduce to a single value.
302
780
  if (config.azure) {
303
781
  forges.azure = { auth: null, host: makeAzureHostFn({ orgUrl: config.azure.orgUrl }) };
304
- try {
305
- forges.azure.auth = await makeAzureAuthFn(config.azure);
306
- log("self_identity", { kind: "azure", id: forges.azure.auth.selfId?.id ?? null, source: forges.azure.auth.source });
307
- } catch (err) {
308
- log("azure_auth_unavailable", { kind: "azure", reason: err?.message });
309
- }
782
+ await attachAuth("azure", makeAzureAuthFn, config.azure, (auth) => auth.selfId?.id ?? null);
310
783
  }
311
784
 
312
785
  /** The `{ auth, host }` pair a job's kind names, or `undefined` for a local job (which has no forge). */
@@ -331,40 +804,97 @@ export async function startWorker(
331
804
  // and a forgotten one is INVISIBLE, because `reapAll` is conservative over the reapers it is handed
332
805
  // rather than over the venues that exist. A bundle already carries its own `reap`, so taking it from
333
806
  // there is one place. `local`'s is built here because its bundle cannot exist yet.
334
- backendReaps = { [DEFAULT_BACKEND]: makeReaperFn({ log }), ...Object.fromEntries(extraBackends.map((b) => [b?.name, b?.reap])) };
335
- reaped = (await reapAll(Object.values(backendReaps), { log }))?.reaped === true;
807
+ // Issue #354: `local`'s only while it is blessed, since its bundle is built only then and the registry refuses a
808
+ // boot reaper for a venue it does not hold.
809
+ // Issue #354: podman's the same way and for the same reason, `makeReaper` with `bin: "podman"`, so the sweep lists
810
+ // the store this account's jobs actually ran in. Docker's listing says nothing about it, and the reverse.
811
+ backendReaps = {
812
+ ...(localBlessed ? { [DEFAULT_BACKEND]: makeReaperFn({ log }) } : {}),
813
+ ...(podmanBlessed ? { [PODMAN_BACKEND]: makePodmanReaperFn({ log }) } : {}),
814
+ ...Object.fromEntries(extraBackends.map((b) => [b?.name, b?.reap])),
815
+ };
816
+ const swept = (await reapAll(Object.values(backendReaps), { log }))?.reaped === true;
817
+ // PROVEN FOR THE WHOLE HOST, or not at all (issue #354). `reapAll` is conservative over the reapers it is
818
+ // handed, and two gaps sit outside that. A BLESSED venue with no reaper here is refused by the registry, but
819
+ // only further down, after the scope sweep has already acted on this answer. And a host without `local` can
820
+ // still hold `pi-job-` containers under Docker, from before `local` was dropped from PI_BACKENDS, that no
821
+ // blessed venue's reaper lists: "every blessed venue enumerated" is then true while this host is not shown
822
+ // clean. The scope sweep is an optimisation over the TTL, so declining it costs one TTL of a stale claim, never
823
+ // a slot; claiming it would free slots for containers that may still be running. A venue that can prove the
824
+ // Docker side too (a read-only listing) is what lifts the second gap, and it is not this change.
825
+ reaped = swept && localBlessed && config.backends.every((name) => typeof backendReaps[name] === "function");
826
+ // Its own event, said once at boot: the sweeper's `scope_claims_sweep_skipped` says only that the reap did not
827
+ // prove the host, and this is the one case where every reaper answered and the host is still not proven.
828
+ if (swept && !reaped) log("host_reap_unproven", { reason: localBlessed ? "a blessed backend has no boot reaper" : "local is not blessed, so no reaper lists this host's docker containers" });
336
829
  } catch (err) {
337
- log("reaper_skipped", { reason: err?.message });
830
+ log("reaper_skipped", { reason: scrubCredentials(err?.message) });
338
831
  }
339
832
 
340
833
  // REQ-LOCAL-JOB-VISIBILITY: sweep aged `.log`/`.json` history at boot so the logs directory stays
341
834
  // bounded across restarts. Best-effort with the same double-wrap posture as the container reaper: the
342
835
  // reaper swallows its own fs errors, and this guard keeps any reaper failure from blocking draining.
836
+ // HELD, not discarded: the periodic sweep (issue #292) re-runs this exact closure, so one configuration
837
+ // read serves boot and every tick after it and the two cannot drift. The `let` with an inert default is
838
+ // what keeps a throwing FACTORY from leaving the sweep holding `undefined` -- the guard below promises
839
+ // that no reaper failure blocks draining, and that promise now has to cover construction too.
840
+ let reapLogs = () => {};
343
841
  try {
344
- await makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log })();
842
+ reapLogs = makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log });
843
+ await reapLogs();
345
844
  } catch (err) {
346
- log("log_reaper_skipped", { reason: err?.message });
845
+ log("log_reaper_skipped", { reason: scrubCredentials(err?.message) });
347
846
  }
348
847
 
349
848
  // REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
350
849
  // kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
351
- // retention policy, a different PII class, and one thing neither sibling needs: it asks docker which
352
- // sandboxes are live first, because an operator's shell can outlive a worker restart by design and
850
+ // retention policy, a different PII class, and one thing neither sibling needs: it asks the container runtimes
851
+ // which sandboxes are live first, because an operator's shell can outlive a worker restart by design and
353
852
  // deleting a bind mount underneath it is a confusing failure with a boring cause. Same double-wrap.
853
+ // Held for the periodic sweep, same as the log reaper above. Safe to re-run on a timer without any
854
+ // state carried between calls: `makeSandboxReaper` declares `running` INSIDE its returned function and
855
+ // re-issues `listRunning` every call, so a docker outage costs one interval of retention overshoot
856
+ // rather than latching the sweep off until the next boot.
857
+ let reapSandboxes = async () => {};
354
858
  try {
355
- await makeSandboxReaperFn({
859
+ // Issue #429, as corrected by its review: which sandboxes are open is asked PER RETAINED RUN, of the runtime that
860
+ // run's manifest records, never of the runtimes this worker blesses. The opener's blessing is the OPENER's
861
+ // `PI_BACKENDS` and routinely differs from the worker's (`OQ-038`), so a blessed-list listing let a podman-only
862
+ // worker delete a local run's directory under a docker shell an operator had just opened. A runtime that cannot
863
+ // answer holds its own runs this pass and nothing else, so a stale docker CLI does not stop a podman sweep, and a
864
+ // run it cannot place (an unreadable manifest, a venue with no launcher) is asked of every runtime present.
865
+ // `blessed` only adds runtimes to the network sweep; a host that blesses neither and retains nothing from either
866
+ // spawns neither CLI.
867
+ // The store a podman run recorded is compared with THIS worker's podman store (review round 2): the cached boot
868
+ // read when podman is blessed, else one read through the same seam, asked only when a podman run recorded one.
869
+ const readPodmanStore = async () => {
870
+ const read = await (podmanInfo ?? readPodmanInfoFn)();
871
+ return read?.answered === true ? (read.info?.graphRoot ?? null) : null;
872
+ };
873
+ const watch = makeSandboxRuntimeWatch({ sandboxDir: config.sandboxDir, blessed: config.backends, list: listRunningSandboxesFn, readPodmanStore, makeSweeper: makeSandboxNetworkSweeperFn, proxy: config.egressProxy, log });
874
+ reapSandboxes = makeSandboxReaperFn({
356
875
  sandboxDir: config.sandboxDir,
357
876
  retentionHours: config.sandboxRetentionHours,
358
- listRunning: listRunningSandboxes,
877
+ listRunning: watch.listRunning,
878
+ // Issue #337: the session networks a died-mid-session process left, one sweeper per runtime present (issue
879
+ // #429), each listing and removing in its own runtime. Unconditional on `PI_EGRESS`, on the boot reaper's own
880
+ // precedent (it lists `pi-job-` networks whatever the posture) and for a sharper reason: a deployment that has
881
+ // turned the policy OFF is exactly where the leftovers are guaranteed dead, since nothing is making new ones.
882
+ sweepNetworks: watch.sweepNetworks,
883
+ // Issue #446, gate round 1: each expired run's own runtime is asked once more right before it is renamed aside.
884
+ isOpen: watch.isOpen,
359
885
  log,
360
- })();
886
+ });
887
+ await reapSandboxes();
361
888
  } catch (err) {
362
- log("sandbox_reaper_skipped", { reason: err?.message });
889
+ log("sandbox_reaper_skipped", { reason: scrubCredentials(err?.message) });
363
890
  }
364
891
 
365
892
  // One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
366
893
  // hoisted out of the createWorkerFn arg object.
367
- const redis = makeRedisClient(config.valkeyUrl);
894
+ const redis = makeRedisClient(valkey.url, { servername: valkey.servername, context: valkeyContext });
895
+ // Its errors as one message-only line (PR #475's review, round 2), like every Queue and Worker's: without a listener
896
+ // ioredis printed a stack per reconnect attempt.
897
+ onValkeyError(redis, "shared client");
368
898
 
369
899
  // This host's own stale scope claims, gated on the reaper having having enumerated: the
370
900
  // reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
@@ -376,14 +906,14 @@ export async function startWorker(
376
906
  if (config.workerNameDeclared)
377
907
  await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current.map((r) => ({ concurrent: r.concurrent, hash: scopeKeyPrefix(r.scope).slice("budget:s:".length) })), log })({ reaped });
378
908
  } catch (err) {
379
- log("scope_claims_sweep_skipped", { reason: err?.message });
909
+ log("scope_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
380
910
  }
381
911
 
382
912
  // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
383
913
  // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
384
914
  // Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
385
915
  // drains it after the worker.
386
- const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
916
+ const runtimeQueue = makeQueue(valkeyConn());
387
917
 
388
918
  // THE HOST QUEUE (issue #57): work only this machine can do, because the folder lives here.
389
919
  //
@@ -399,7 +929,7 @@ export async function startWorker(
399
929
  // live triggers-file edit lands on the same queue the boot reconcile used; otherwise the shared
400
930
  // runtime queue, exactly as before. Registered as an extraCloser only when it is a NEW handle --
401
931
  // closing `runtimeQueue` twice would be closing another owner's connection.
402
- const cronQueue = hostQueue ? makeQueue(parseConnection(config.valkeyUrl), { name: hostQueue }) : runtimeQueue;
932
+ const cronQueue = hostQueue ? makeQueue(valkeyConn(), { name: hostQueue }) : runtimeQueue;
403
933
 
404
934
  // REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
405
935
  // captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
@@ -418,15 +948,19 @@ export async function startWorker(
418
948
  maxAgeDays: config.sessionMaxAgeDays,
419
949
  maxResumeChain: config.sessionMaxResumeChain,
420
950
  maxContextPct: config.sessionMaxContextPct,
951
+ // The venue a transcript is stamped with and gated on (#277), resolved with the registry's own default.
952
+ defaultBackend: config.defaultBackend,
421
953
  log,
422
954
  });
423
955
  // Boot sweep, beside the log reaper and for the same reason it is beside rather than inside it: these
424
- // files have a different retention policy and a different PII class. The gate that actually matters is
425
- // the age check at OPEN -- a worker that never restarts would otherwise resume forever (OQ-007).
956
+ // files have a different retention policy and a different PII class. Until issue #292 the age check at
957
+ // OPEN was carrying this alone, because a worker that never restarts never re-swept; the periodic sweep
958
+ // now covers the DISK half and the open gate covers the INPUT half, which is the one that matters
959
+ // (OQ-007, RESOLVED).
426
960
  try {
427
961
  sessionStore.reapSessions();
428
962
  } catch (err) {
429
- log("session_reaper_skipped", { reason: err?.message });
963
+ log("session_reaper_skipped", { reason: scrubCredentials(err?.message) });
430
964
  }
431
965
  // The one-shot file path (issue #231): PI_TRIGGERS_FILE, else ./triggers.json against this process's
432
966
  // cwd -- doctor's own fallback, chosen for doctor's own reason ("the two must read the same file"),
@@ -443,7 +977,9 @@ export async function startWorker(
443
977
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
444
978
  // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
445
979
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
446
- const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName });
980
+ // The default venue rides the same way and for the same reason (#277): it is the value the registry
981
+ // below is built with, so the record resolves a job's venue exactly as dispatch does.
982
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend });
447
983
  writeRecord(record);
448
984
  // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
449
985
  // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
@@ -555,16 +1091,102 @@ export async function startWorker(
555
1091
  // ONE preflight instance, constructed once and shared: `start-wiring.test.mjs` pins that, and the
556
1092
  // reason is the module's own -- the tag the preflight checked has to be the tag `docker run` is
557
1093
  // handed, and two constructions are two chances for that to stop being true.
558
- const imagePreflight = makeImagePreflightFn({ image: config.jobImage });
1094
+ // Issue #354: built only while `local` is blessed, since it IS local's (`docker image inspect`). Without `local` the
1095
+ // boot read below asks the default venue's own preflight instead, the one its jobs will be gated on.
1096
+ const imagePreflight = localBlessed ? makeImagePreflightFn({ image: config.jobImage }) : null;
1097
+ // Issue #354: the podman bundle, built only while `podman` is blessed, from the SAME deployment inputs local's
1098
+ // runContainer is handed below (one image, one overlay, one forward list, one egress posture: a venue must not quietly
1099
+ // run a different job than the one configured), plus what only this venue reads: the cached `podman info` the boot
1100
+ // read above already asked, the floor its per-job observation preflight judges, the boot-built reaper, and the
1101
+ // process's identity and host files, through the same seams local's decisions read. Built HERE rather than beside
1102
+ // local's, because the boot image read just below asks the default venue's own preflight when that venue is podman.
1103
+ const podmanBackend = podmanBlessed
1104
+ ? makePodmanBackendFn({
1105
+ image: config.jobImage,
1106
+ hostEnv: env,
1107
+ egress: config.egress,
1108
+ egressProxy: config.egressProxy,
1109
+ openJobLog,
1110
+ globalPiDir: config.globalPiDir,
1111
+ allowGlobalExtensions: config.allowGlobalExtensions,
1112
+ packagePaths: getPackagePaths,
1113
+ forwardEnv: config.forwardEnv,
1114
+ authFromPi: config.authFromPi,
1115
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
1116
+ backendFloor: config.backendFloor,
1117
+ reap: backendReaps[PODMAN_BACKEND],
1118
+ readInfo: podmanInfo,
1119
+ platform: jobUserIdentity.platform,
1120
+ euid: jobUserIdentity.euid,
1121
+ egid: jobUserIdentity.egid,
1122
+ fs: observationFs,
1123
+ home: jobUserIdentity.home,
1124
+ env,
1125
+ log,
1126
+ })
1127
+ : null;
1128
+ // Issue #458 (PR #463 round 2): on Podman 4.x with egress armed, the keeper read ONCE at boot through the bundle's own
1129
+ // preflight, bounded, and said as its own event with the whole sentence. A deployment that upgraded without re-running
1130
+ // `service install` or `up` has no keeper, and would otherwise learn it only from retried jobs. Only a warning: the
1131
+ // per-job preflight is the gate: once the keeper holds, the next job runs (under a proxy up longer than the grace,
1132
+ // 15 s, after the proxy's restart, which the job's retry message and doctor both ask for).
1133
+ if (podmanBackend && config.egress) {
1134
+ const readBootKeeper = () =>
1135
+ settleWithin(
1136
+ Promise.resolve()
1137
+ .then(() => podmanBackend.egressPreflight({}))
1138
+ .catch(() => ({})),
1139
+ BOOT_IMAGE_TIMEOUT_MS * 4,
1140
+ {},
1141
+ );
1142
+ let bootKeeper = await readBootKeeper();
1143
+ // Issue #476: a keeper whose only fault is its age is waited out, once and bounded (at most the minimum age plus
1144
+ // the margin), then judged again. A stack started together reads it under a second old (measured 0.5 to 0.8 s on
1145
+ // 4.9.3), and warning then was wrong every time; a keeper still young after the wait restarted in it, and is said.
1146
+ if (bootKeeper?.young && typeof bootKeeper.young === "object") {
1147
+ const waitMs = Math.min(Number.isFinite(bootKeeper.young.waitMs) ? bootKeeper.young.waitMs : 0, NETNS_KEEPER_MIN_AGE_MS + NETNS_KEEPER_YOUNG_MARGIN_MS);
1148
+ // Said as information, not a warning: the one line that shows a joint start was waited out, and for how long.
1149
+ log("netns_keeper_young_at_boot", { keeperAgeMs: bootKeeper.young.ageMs ?? null, waitMs });
1150
+ await sleep(Math.max(0, waitMs));
1151
+ bootKeeper = await readBootKeeper();
1152
+ }
1153
+ // Its own sentence (PR #463 round 3): the job-shaped one says "this job is retried", and at boot there is no job.
1154
+ if (typeof bootKeeper?.keeper === "string") log("netns_keeper_not_holding_at_boot", { reason: bootKeeper.keeperAtBoot ?? bootKeeper.keeper });
1155
+ }
1156
+ // Every bundle this boot built beside local's, in registration order: the podman one first, then the injected ones.
1157
+ const builtBackends = [...(podmanBackend ? [podmanBackend] : []), ...extraBackends];
559
1158
  // BOUNDED, because `.catch()` cannot rescue a promise that never settles: `runDocker` resolves only on
560
1159
  // the child's `close` or `error` and has no timeout of its own, so a wedged daemon would hang boot
561
1160
  // here. This read is a nicety -- a digest for the boot line and the registry -- and a nicety may
562
1161
  // never be able to stop a worker starting. The per-JOB preflight keeps its unbounded wait, where a
563
1162
  // wedged daemon is the job's problem and the 30-minute job timeout already covers it.
564
- const bootImage = await Promise.race([
565
- imagePreflight({}).catch(() => ({})),
566
- new Promise((resolve) => setTimeout(() => resolve({}), BOOT_IMAGE_TIMEOUT_MS).unref?.()),
567
- ]);
1163
+ // Without `local` (issue #354) the default venue's own preflight, under the same bound and the same swallow: a
1164
+ // nicety that throws synchronously must not stop a boot either, hence the `then`. None at all reads as no digest.
1165
+ const defaultVenueImageRead = () => {
1166
+ const read = builtBackends.find((b) => b?.name === config.defaultBackend)?.imagePreflight;
1167
+ return Promise.resolve()
1168
+ .then(() => (typeof read === "function" ? read({}) : {}))
1169
+ .then((answer) => answer ?? {})
1170
+ .catch(() => ({}));
1171
+ };
1172
+ const bootImage = await settleWithin(imagePreflight ? imagePreflight({}).catch(() => ({})) : defaultVenueImageRead(), BOOT_IMAGE_TIMEOUT_MS, {});
1173
+ // Issue #341: say it at boot, not only as a refusal on every job. The deployment's default image cannot run as
1174
+ // this worker's uid on this daemon, so every local job that uses it will be refused pre-spend.
1175
+ if (bootDecision?.mode === "worker" && bootImage.ok) {
1176
+ const planned = resolveImageUser(bootDecision, { capabilities: bootImage.capabilities ?? [], euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, socket: bootJobUser?.socket ?? null });
1177
+ if (planned.refused === "job-image-any-uid-unsupported") log("job_image_any_uid_unsupported", { image: config.jobImage });
1178
+ else if (planned.refused) log("job_user_group_refused", { cause: planned.cause });
1179
+ if (planned.user && config.forwardEnv.includes("HOME")) log("forward_env_home_overridden", { reason: "HOME is set beside --user" });
1180
+ }
1181
+ // Issue #354: the same boot sentence for a podman DEFAULT venue, whose image was read from this account's own store
1182
+ // just above. Only as the default: with `local` the default, `bootImage` is docker's copy of the tag, which says
1183
+ // nothing about the one in the podman store. `backend` names the venue, since `local`'s lines carry none.
1184
+ if (config.defaultBackend === PODMAN_BACKEND && bootPodmanDecision?.mode === "worker" && bootImage.ok) {
1185
+ const planned = resolvePodmanImageUser(bootPodmanDecision, { capabilities: bootImage.capabilities ?? [], euid: jobUserIdentity.euid, egid: jobUserIdentity.egid });
1186
+ if (planned.refused === "job-image-any-uid-unsupported") log("job_image_any_uid_unsupported", { image: config.jobImage, backend: PODMAN_BACKEND });
1187
+ else if (planned.refused) log("job_user_group_refused", { cause: planned.cause, backend: PODMAN_BACKEND });
1188
+ if (planned.user && config.forwardEnv.includes("HOME")) log("forward_env_home_overridden", { reason: "HOME is set beside --user", backend: PODMAN_BACKEND });
1189
+ }
568
1190
  // Resolved once: `Intl` is not free, and this value cannot change without a restart.
569
1191
  const hostTz = Intl.DateTimeFormat().resolvedOptions().timeZone ?? "";
570
1192
  const registry = makeHostRegistryFn({ redis, name: config.workerName, log });
@@ -606,6 +1228,107 @@ export async function startWorker(
606
1228
  });
607
1229
 
608
1230
 
1231
+ // `local`'s two optional preflights (issue #354: they were `deps` closures, and are now the bundle's own members,
1232
+ // dispatched per venue by the registry). Bodies unchanged.
1233
+ //
1234
+ // Issue #278: the docker endpoint read AGAIN before each job's spend, because a `docker context use`
1235
+ // after boot redirects every later job, and a preflight that answered once at boot would give a
1236
+ // wrong decision all day (the image preflight is not cached for the same reason). Only for a venue
1237
+ // whose declaration is observation-gated on it, and only a refusal under a floor that needs it.
1238
+ // Issue #345: the runtime observations come from the same per-job read, in the same place: the facts the job user is
1239
+ // decided from (cached per endpoint state, so no second daemon call) and this host's Podman files, re-read per job
1240
+ // because an operator creating the empty mounts.conf override must not need a restart.
1241
+ const localObservationPreflight = async (job) => {
1242
+ const venue = resolveBackendName(job, config.defaultBackend);
1243
+ if (Object.keys(backendFor(venue)?.observedBy ?? {}).length === 0) return { ok: true };
1244
+ const endpoint = await resolveDockerEndpointFn();
1245
+ const state = dockerEndpointState(endpoint);
1246
+ if (state !== endpointSeen) {
1247
+ endpointSeen = state;
1248
+ logDockerEndpoint(log, endpoint, { changed: true });
1249
+ }
1250
+ // The endpoint refusal FIRST, from the CLI's own configuration only, as boot does: a floor that distrusts this
1251
+ // endpoint must not wait on, or send the CLI's own TLS client credentials to, the daemon behind it.
1252
+ const endpointArgs = {
1253
+ backends: [venue],
1254
+ backendFloor: config.backendFloor,
1255
+ observations: { [DOCKER_ENDPOINT_LOCAL]: endpoint.local === true },
1256
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(endpoint) },
1257
+ only: [DOCKER_ENDPOINT_LOCAL],
1258
+ };
1259
+ const [endpointRefusal] = observationRefusals(endpointArgs);
1260
+ if (endpointRefusal) {
1261
+ if (endpoint.local === null && endpoint.transient) return { unavailable: true, reason: endpoint.reason };
1262
+ return { refused: true, message: endpointRefusal, observations: [DOCKER_ENDPOINT_LOCAL] };
1263
+ }
1264
+ const jobUser = await resolveJobUser({ endpoint, key: state });
1265
+ const unit = await readRootfulService({ endpoint, daemon: jobUser.daemon, readService: readPodmanServiceFn });
1266
+ // One read of each host path for this job's two checks (`onceFs`, gate round 3 of PR #473), fresh per job.
1267
+ const jobFs = onceFs(observationFs);
1268
+ const observed = observeHost({ endpoint, daemon: jobUser.daemon, fs: jobFs, unit, env, memory: rootfulMemory });
1269
+ if (runtimeObservationKey(observed) !== runtimeObservedSaid) {
1270
+ runtimeObservedSaid = runtimeObservationKey(observed);
1271
+ log("runtime_observed", { daemonAppliesBounds: observed.observations.daemonAppliesBounds, runtimeAddsNoMounts: observed.observations.runtimeAddsNoMounts, changed: true });
1272
+ }
1273
+ // Issue #448: rootful Podman's containers.conf, per job and before any spend, re-read every time because removing a
1274
+ // key must need no worker restart (only the Podman service's). Handed back as the podman venue's refusal is
1275
+ // (`podmanConfRefused`, `rootful: true`), ahead of the floor: it is what the venue IS on this host, and the processor
1276
+ // returns it before the image preflight. Nothing is read where the daemon is not rootful Podman on this host.
1277
+ const rootful = await observeRootfulConf({ endpoint, daemon: jobUser.daemon, fs: jobFs, readService: readPodmanServiceFn, env, unit, memory: rootfulMemory });
1278
+ sayRootfulUnread(rootful);
1279
+ if (rootful?.refusal) return { ok: true, endpoint, jobUser, podmanConfRefused: rootfulRefused(rootful.refusal) };
1280
+ const args = {
1281
+ backends: [venue],
1282
+ backendFloor: config.backendFloor,
1283
+ observations: observed.observations,
1284
+ evidence: { [DOCKER_ENDPOINT_LOCAL]: dockerEndpointEvidence(endpoint), ...observed.evidence },
1285
+ };
1286
+ const [refusal] = observationRefusals(args);
1287
+ if (!refusal) return { ok: true, endpoint, jobUser };
1288
+ const missed = [...new Set(unobservedFloor(args.backends, args.backendFloor, args.observations).map((m) => m.observedBy))];
1289
+ if (observationRefusalIsTransient(args)) return unavailableFor(observed, missed[0]);
1290
+ return { refused: true, message: refusal, observations: missed };
1291
+ };
1292
+ // Issue #452, gate round 5: the runtime each local job was admitted on, by job id, taken (and forgotten) by that job's
1293
+ // teardown. Bounded: a job refused after its job-user preflight never reaches a teardown, so the oldest entries go first.
1294
+ const admittedRuntimes = new Map();
1295
+ const ADMITTED_RUNTIMES_MAX = 1000;
1296
+ const recordAdmittedRuntime = (job, runtime) => {
1297
+ if (job?.id === undefined || runtime === undefined) return;
1298
+ admittedRuntimes.delete(job.id);
1299
+ admittedRuntimes.set(job.id, runtime);
1300
+ while (admittedRuntimes.size > ADMITTED_RUNTIMES_MAX) admittedRuntimes.delete(admittedRuntimes.keys().next().value);
1301
+ };
1302
+ const takeAdmittedRuntime = (job) => {
1303
+ const runtime = admittedRuntimes.get(job?.id);
1304
+ admittedRuntimes.delete(job?.id);
1305
+ return runtime;
1306
+ };
1307
+ // Issue #341: the job user, for a job on the `local` venue only (its containers are this host's docker
1308
+ // CLI's). The endpoint is the one `observationPreflight` just read, so one job's two decisions agree.
1309
+ const localJobUserPreflight = async (job, { capabilities = [], observed } = {}) => {
1310
+ const venue = resolveBackendName(job, config.defaultBackend);
1311
+ if (venue !== DEFAULT_BACKEND) return { user: null, home: null };
1312
+ const endpoint = observed?.endpoint ?? (await resolveDockerEndpointFn());
1313
+ const admittedOn = observed?.jobUser ?? (await resolveJobUser({ endpoint, key: dockerEndpointState(endpoint) }));
1314
+ const { decision, socket, facts } = admittedOn;
1315
+ // Issue #452, gate round 5: the runtime THIS job is admitted on, recorded per job for its teardown's detach gate. The
1316
+ // resolver's cache is per endpoint state and moves when the endpoint does, so reading it at teardown could hand a
1317
+ // job the answer of another endpoint's daemon.
1318
+ recordAdmittedRuntime(job, admittedOn?.daemon?.answered ? runtimeFromFacts(admittedOn.daemon) : undefined);
1319
+ if (jobUserLogKey(decision) !== jobUserSaid) {
1320
+ jobUserSaid = jobUserLogKey(decision);
1321
+ log("job_user", { mode: decision.mode, user: decision.user, cause: decision.cause, reason: decision.reason });
1322
+ }
1323
+ const chosen = resolveImageUser(decision, { capabilities, euid: jobUserIdentity.euid, egid: jobUserIdentity.egid, socket });
1324
+ // Issue #355: whether this job's own mounts carry `:Z`, from the SAME facts and endpoint the user was decided
1325
+ // from, so one job's two answers cannot come from two reads. Only on a path that runs (a refusal or an
1326
+ // undecidable daemon runs nothing), and only when true: every host this does not apply to keeps the answer
1327
+ // shape, and so the argv, it had before.
1328
+ if (chosen.refused || chosen.unavailable) return chosen;
1329
+ return relabelsPrivateMounts(facts, endpoint, jobUserIdentity.platform) ? { ...chosen, relabel: true } : chosen;
1330
+ };
1331
+
609
1332
  // #227: WHERE this job's container runs. The three functions that decide whether a container may start and
610
1333
  // then start it -- two pre-spend gates and the launcher -- bundled into one value with a completeness
611
1334
  // check, so the set has a name instead of being three unrelated `deps` keys. Byte-identical to passing them individually -- the same three functions reach the same three keys,
@@ -615,70 +1338,193 @@ export async function startWorker(
615
1338
  // Assigned key by key below rather than spread, because the bundle also carries `name` and `declares`,
616
1339
  // and `deps` is the processor's namespace: a spread would put a backend's name into it under a key the
617
1340
  // processor is free to mean something else by.
618
- const localBackend = makeLocalBackend({
619
- // #227. The two the earlier slices deferred, now real seams. `reap` keeps its tri-state: the boot
620
- // sweep below only sweeps this host's scope claims once the reaper has PROVEN this host holds no
621
- // job containers, and an unproven answer must never free a slot.
622
- stopContainer: makeStopContainer(),
623
- reap: backendReaps[DEFAULT_BACKEND],
624
- // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
625
- // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
626
- // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
627
- // Nothing is memoised (see its construction above): `docker image inspect` costs ~tens of ms against a
628
- // container run of minutes, and a cache would be wrong in both directions -- an operator who builds the
629
- // image mid-day would stay refused, one who removes it would stay admitted. Contrast the staged-package
630
- // manifest, correctly read once at boot because it is deploy-time state under a :ro mount.
631
- imagePreflight,
632
- // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
633
- // value, one place, so the gate that checks the proxy and the runner that attaches to its network
634
- // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
635
- // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
636
- // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
637
- egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
638
- runContainer: makeRunContainerFn({
639
- image: config.jobImage,
640
- hostEnv: env,
641
- egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
642
- egressProxy: config.egressProxy,
643
- openJobLog,
644
- globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
645
- allowGlobalExtensions: config.allowGlobalExtensions,
646
- // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
647
- // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
648
- // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
649
- packagePaths: getPackagePaths,
650
- forwardEnv: config.forwardEnv,
651
- authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
652
- // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
653
- // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
654
- // adding one does not widen this signature again.
655
- forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
656
- }),
657
- });
1341
+ //
1342
+ // Issue #354: built ONLY while `local` is blessed. Built unconditionally, as it was while PI_BACKENDS had to include
1343
+ // it, it would be a registered venue this deployment never chose, whose reaper the boot map must then carry and
1344
+ // whose reads spawn `docker` on a host that may have none. The two optional preflights ride on the bundle rather
1345
+ // than through `makeLocalBackend`, whose completeness check treats every member it takes as required.
1346
+ const localBackend = localBlessed
1347
+ ? {
1348
+ ...makeLocalBackend({
1349
+ // #227. The two the earlier slices deferred, now real seams. `reap` keeps its tri-state: the boot
1350
+ // sweep below only sweeps this host's scope claims once the reaper has PROVEN this host holds no
1351
+ // job containers, and an unproven answer must never free a slot.
1352
+ stopContainer: makeStopContainer(),
1353
+ reap: backendReaps[DEFAULT_BACKEND],
1354
+ // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
1355
+ // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
1356
+ // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
1357
+ // Nothing is memoised (see its construction above): `docker image inspect` costs ~tens of ms against a
1358
+ // container run of minutes, and a cache would be wrong in both directions -- an operator who builds the
1359
+ // image mid-day would stay refused, one who removes it would stay admitted. Contrast the staged-package
1360
+ // manifest, correctly read once at boot because it is deploy-time state under a :ro mount.
1361
+ imagePreflight,
1362
+ // REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
1363
+ // value, one place, so the gate that checks the proxy and the runner that attaches to its network
1364
+ // cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
1365
+ // starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
1366
+ // Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
1367
+ egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
1368
+ runContainer: makeRunContainerFn({
1369
+ image: config.jobImage,
1370
+ hostEnv: env,
1371
+ egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
1372
+ egressProxy: config.egressProxy,
1373
+ openJobLog,
1374
+ globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
1375
+ allowGlobalExtensions: config.allowGlobalExtensions,
1376
+ // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set
1377
+ // packages:false. A RESOLVER, not the array: the factory is still constructed exactly once, only
1378
+ // the value it reads became a call, so a re-stage takes effect on the next job without a restart.
1379
+ packagePaths: getPackagePaths,
1380
+ forwardEnv: config.forwardEnv,
1381
+ authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
1382
+ // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
1383
+ // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
1384
+ // adding one does not widen this signature again.
1385
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
1386
+ // Issue #452, gate round 4: the teardown's detach gate uses the `docker info` facts this job was admitted
1387
+ // on (the job-user resolver's cached answer), never a fresh read, which on Docker Engine read as
1388
+ // `runtime-unreadable` whenever it timed out or failed and leaked the job's network. No cached answer: it
1389
+ // reads. A refused teardown is logged with its token.
1390
+ teardownRuntime: (job) => takeAdmittedRuntime(job),
1391
+ log,
1392
+ }),
1393
+ }),
1394
+ observationPreflight: localObservationPreflight,
1395
+ jobUserPreflight: localJobUserPreflight,
1396
+ }
1397
+ : null;
658
1398
 
659
1399
  // #227. WHICH backend runs which job, and the one place that decides. One bundle today, so every
660
1400
  // resolution returns it -- but the mechanism is real, so `run.backend` stops being a validated label and
661
1401
  // the abort path can reach a venue it did not build. `config.backends[0]` is the deployment's default,
662
1402
  // and the registry refuses a default it does not hold rather than discovering it at the first pickup.
663
- const backends = makeBackendRegistryFn({
664
- // #227. A SEAM, not a literal. `docs/backends.md` tells an adapter author to register their bundle
665
- // here, and until this was injectable that instruction described code nobody could run: the array
666
- // was hard-coded, so a venue could pass the conformance suite, get a table entry and be blessed in
667
- // PI_BACKENDS, and then be refused at boot as blessed-but-unbuilt with nowhere to put it. It is also
668
- // what lets a wiring test prove `startWorker` actually CONNECTS the registry to the processor --
669
- // six mutations reverting that connection survived the whole suite, which is the same shape as the
670
- // bug that shipped: invisible while there is one venue.
671
- bundles: [localBackend, ...extraBackends],
672
- defaultName: config.defaultBackend,
673
- // Cross-checked at boot rather than discovered at the first pickup: a name PI_BACKENDS blesses but
674
- // nothing builds passes both the loader and the pre-spend gate, and a venue with no boot reaper is
675
- // swept by nothing while still reporting the host as proven clean.
676
- blessed: config.backends,
677
- reaps: backendReaps,
1403
+ // A refusal here comes AFTER boot opened the Redis client, the runtime and cron queues and the host registry, so all of
1404
+ // them are released before the refusal travels: any one left open keeps the event loop alive, and a worker that refused
1405
+ // to boot would hang instead of exiting (measured against a real Valkey: two sockets held past 30 s).
1406
+ let backends;
1407
+ try {
1408
+ backends = makeBackendRegistryFn({
1409
+ // #227. A SEAM, not a literal. `docs/backends.md` tells an adapter author to register their bundle
1410
+ // here, and until this was injectable that instruction described code nobody could run: the array
1411
+ // was hard-coded, so a venue could pass the conformance suite, get a table entry and be blessed in
1412
+ // PI_BACKENDS, and then be refused at boot as blessed-but-unbuilt with nowhere to put it. It is also
1413
+ // what lets a wiring test prove `startWorker` actually CONNECTS the registry to the processor --
1414
+ // six mutations reverting that connection survived the whole suite, which is the same shape as the
1415
+ // bug that shipped: invisible while there is one venue.
1416
+ bundles: [...(localBackend ? [localBackend] : []), ...builtBackends],
1417
+ defaultName: config.defaultBackend,
1418
+ // Cross-checked at boot rather than discovered at the first pickup: a name PI_BACKENDS blesses but
1419
+ // nothing builds passes both the loader and the pre-spend gate, and a venue with no boot reaper is
1420
+ // swept by nothing while still reporting the host as proven clean.
1421
+ blessed: config.backends,
1422
+ reaps: backendReaps,
1423
+ });
1424
+ } catch (err) {
1425
+ const opened = [registry, runtimeQueue, ...(cronQueue !== runtimeQueue ? [cronQueue] : [])];
1426
+ // Bounded, then forced: a queue whose connection came up closes by sending QUIT and awaiting the reply, and against a
1427
+ // server that stopped answering that wait never ends (measured through a stalling proxy), so the refusal itself would
1428
+ // never travel. Five seconds covers the host registry's own bounded close; whatever is still open is disconnected.
1429
+ await settleWithin(Promise.allSettled(opened.map((handle) => Promise.resolve().then(() => handle?.close?.()))), 5_000);
1430
+ for (const handle of opened) {
1431
+ try {
1432
+ handle?.disconnect?.();
1433
+ } catch {
1434
+ // a handle with nothing left to drop
1435
+ }
1436
+ }
1437
+ redis.disconnect();
1438
+ throw err;
1439
+ }
1440
+
1441
+ // The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
1442
+ // array rather than the literal it used to be, because up to three of its members do not exist yet: the
1443
+ // live-edit watchers are armed at the END of boot, below, and deliberately after the boot reconcile --
1444
+ // arming them earlier would let an operator edit run `reloadSchedules` concurrently with the boot
1445
+ // `reconcileGated`, on a different queue handle, and reconcile's orphan prune is not safe against that.
1446
+ //
1447
+ // PUSHING AFTER THE HANDOFF IS SOUND FOR ONE REASON ONLY: `index.mjs` reads this array at SHUTDOWN time,
1448
+ // not when it receives it, and so does the test harness at teardown. A refactor that COPIES it there --
1449
+ // a spread, a freeze, a snapshot inside `createWorker` -- un-registers the watchers in SILENCE and puts
1450
+ // issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
1451
+ const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
1452
+
1453
+ // HOISTED out of the deps literal (issue #288): the terminal failed listener below needs the same
1454
+ // adapter the processor gets, and two bodies would drift exactly where drift costs a public comment.
1455
+ const comment = async (job, text) => {
1456
+ // Best-effort: the processor awaits comment() inside its try, so a rejection here would
1457
+ // corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
1458
+ // This adapter NEVER throws.
1459
+ const forge = forgeFor(job);
1460
+ // Same re-resolve as the mint path, and it matters more here: a comment is how a refusal
1461
+ // reaches the person who asked for the job, so a forge whose auth was merely unreachable at
1462
+ // boot must not degrade every later comment to a stdout line nobody is watching.
1463
+ //
1464
+ // A THROWING re-resolve is treated as no auth rather than as a failed comment, which is what
1465
+ // keeps the fallthrough below reachable: the text still lands on stdout instead of being
1466
+ // replaced by a `comment_failed` line that does not carry it. This adapter never throws.
1467
+ let auth = forge?.auth ?? null;
1468
+ if (forge && !auth) {
1469
+ try {
1470
+ auth = await ensureAuth(job.kind);
1471
+ } catch {
1472
+ auth = null;
1473
+ }
1474
+ }
1475
+ if (auth) {
1476
+ try {
1477
+ const token = await auth.mintToken(job);
1478
+ await forge.host.postStatusComment(job, job.target, text, token);
1479
+ } catch (err) {
1480
+ log("comment_failed", { jobId: job?.id, reason: err?.message });
1481
+ }
1482
+ return;
1483
+ }
1484
+ // A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
1485
+ // post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
1486
+ log("comment", { jobId: job?.id, text });
1487
+ };
1488
+
1489
+ // The operator's failure hook (issue #288, INT-ON-FAILURE-HOOK-CONTRACT). Constructed ONLY when the
1490
+ // knob is set: an unset PI_ON_FAILURE builds nothing, spawns nothing, and logs nothing -- the
1491
+ // byte-identical guarantee. Fired from the two terminal listeners below, never from the processor:
1492
+ // a hook fault must not be able to flip an outcome, and the listeners sit outside every try that
1493
+ // decides one.
1494
+ // `hostEnv: env`, not the default process.env -- the #309 rule every spawner in this file follows
1495
+ // (runContainer, resolveSecrets): a subprocess runs with THIS worker's env, identical on the real
1496
+ // path and divergent only under an injected one, which is exactly where the difference would hide.
1497
+ const onFailure = config.onFailure
1498
+ ? makeOnFailure({ command: config.onFailure, timeoutMs: config.onFailureTimeoutMs, host: config.workerName ?? "", hostEnv: env, log })
1499
+ : null;
1500
+ // Which POLICY reasons page the operator. Paid terminals only: worker-abort, runner-policy and every
1501
+ // named runner reason cost a container and ended wrong. The runner's named reasons are SPREAD from
1502
+ // RUNNER_POLICY_REASONS rather than listed, because each is an exit 2 that would have paged as
1503
+ // runner-policy before it got its own label, and a label must not be the thing that silences a page;
1504
+ // provider-auth-refused (issue #437) is the case in point, a refusal only the operator can fix. Excluded on purpose: `completed` and every pre-spend refusal (free, and
1505
+ // each already comments -- a delivery storm against a spent cap must not page anyone), and
1506
+ // `operator-cancel`, because the operator initiated it and a push telling them what they just did is
1507
+ // noise with a pager attached.
1508
+ const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", ...RUNNER_POLICY_REASONS]);
1509
+ // The infra-terminal sentence (issue #288). FIXED, never err.message: the message classes that reach
1510
+ // a failedReason carry host paths and library words (the #310 record), and for a local job this text
1511
+ // lands verbatim in the service log through the adapter's stdout fallthrough. The worker log already
1512
+ // holds the truncated reason on the job_failed line beside it.
1513
+ const FAILED_COMMENT = "Failed: an error stopped this job and it will not be retried further. Ask the operator to check the worker log.";
1514
+ // Issue #458 (PR #463 round 2): a job that never started because the podman venue's rootless network keeper did not
1515
+ // hold says so, since the generic line sends an operator to a log that only says the job failed. Fixed text keyed by
1516
+ // the fixed reason token, never the error's own words: a forge comment carries nothing a host read produced.
1517
+ const FAILED_COMMENT_BY_REASON = Object.freeze({
1518
+ // Neutral on the cause (PR #463 round 3): a keeper that is not running and a proxy that must restart after it both land here.
1519
+ [NETNS_KEEPER_NOT_HOLDING]: "Failed before it started: this worker's rootless Podman egress proxy did not pass its pre-start check (its rootless network keeper, pi-dispatch-netns-keeper), so the job was retried and never run. Nothing was spent. Ask the operator to run `pi-dispatch doctor` on the worker for the exact fix.",
1520
+ // Issue #476: a job held on a young keeper that kept restarting while it waited.
1521
+ [NETNS_KEEPER_CRASH_LOOP]: "Failed before it started: this worker's rootless network keeper (pi-dispatch-netns-keeper), which its Podman egress proxy needs, kept restarting while the job waited for it, so the job was retried and never run. Nothing was spent. Ask the operator to run `pi-dispatch doctor` on the worker and read the keeper's log (`journalctl --user -u pi-dispatch-netns-keeper.service`).",
1522
+ // Issue #448 (gate round 2 of PR #473): the hold's own ending, so the comment names the cause and the fix.
1523
+ [PODMAN_RESTART_HOLD_EXPIRED]: "Failed before it started: rootful Podman's service on this worker kept running with a containers.conf older than its files (or a file's change time stayed ahead of the host's clock) for an hour, so the job was held and never run. Nothing was spent. Ask the operator to restart it while no local job runs (`sudo systemctl restart podman.service`), then run the job again.",
678
1524
  });
679
1525
 
680
1526
  const worker = createWorkerFn({
681
- connection: parseConnection(config.valkeyUrl),
1527
+ connection: valkeyConn(),
682
1528
  // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
683
1529
  // not enough to find the runtime holding it once there is more than one venue.
684
1530
  stopContainer: backends.stopContainer,
@@ -696,7 +1542,7 @@ export async function startWorker(
696
1542
  getSettings,
697
1543
  redis,
698
1544
  recordRun,
699
- extraClosers: [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])],
1545
+ extraClosers,
700
1546
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
701
1547
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
702
1548
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
@@ -743,6 +1589,29 @@ export async function startWorker(
743
1589
  // spending delivery is excused). In the compose topology this check is the once-enforcement
744
1590
  // layer, because the receiver's single-file :ro mount pins a dead inode until restart.
745
1591
  checkOnceSpent: makeCheckOnceSpent({ triggersPath: onceTriggersFile }),
1592
+ // Issue #310. The free provider-credential gate, bound from EXACTLY the values the container builder
1593
+ // is handed, and asserted to be the same by a wiring test. A gate that resolves against different
1594
+ // inputs than the writer is the divergence class this whole cluster of issues is about: it would pass
1595
+ // a job the container then refuses, or refuse one the container would have run. `agentDir` is
1596
+ // defaulted by both, from the same `hostEnv`.
1597
+ //
1598
+ // A processor dep and NOT a backend bundle member: the bundle is a closed set about how a venue runs
1599
+ // a container, and this is a question about the deployment, asked before any venue is chosen.
1600
+ //
1601
+ // A PROBE: whatever it resolves is dropped on the floor. The credential itself is read where it always
1602
+ // was, inside buildContainerEnv, so no live key is ever in scope in the processor.
1603
+ checkProviderCredential: (job) => {
1604
+ try {
1605
+ resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv });
1606
+ return { ok: true };
1607
+ } catch (error) {
1608
+ // Only OUR determinate refusal. Anything else (a bug here, an fs fault the module does not model)
1609
+ // must not become a policy refusal on the operator's issue: it rethrows into runJob's catch, which
1610
+ // classifies it the way it always did.
1611
+ if (error?.piDispatchConfig !== true) throw error;
1612
+ return { ok: false, message: error.message };
1613
+ }
1614
+ },
746
1615
  // Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
747
1616
  // asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
748
1617
  // service below the version floor turns a wait into a paid run nothing can tell from a correct
@@ -774,6 +1643,11 @@ export async function startWorker(
774
1643
  },
775
1644
  imagePreflight: backends.imagePreflight,
776
1645
  egressPreflight: backends.egressPreflight,
1646
+ // Issue #354: both are the VENUE's, dispatched through the registry like the two gates above, so a job on a
1647
+ // venue that carries neither gets the processor's own defaults and never asks this host's docker CLI anything.
1648
+ // `local`'s are the closures built beside its bundle below, unchanged.
1649
+ observationPreflight: backends.observationPreflight,
1650
+ jobUserPreflight: backends.jobUserPreflight,
777
1651
  // Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
778
1652
  // retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
779
1653
  promoteSession: sessionStore.promoteSession,
@@ -789,7 +1663,12 @@ export async function startWorker(
789
1663
  // because the settings file is read at each job start and an operator who declares a profile in
790
1664
  // the panel should not have to restart the worker to use it. A deployment that declares nothing
791
1665
  // spawns nothing at all: the gate only calls this when a trigger is armed.
792
- resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, log }),
1666
+ // `hostEnv` is the env THIS worker was started with, not `process.env` by default (issue #309). It is
1667
+ // the environment the resolver subprocess runs in, and makeRunContainer above is handed the same
1668
+ // `env` for the container it builds. Identical on the real path, where both are process.env; under an
1669
+ // injected env they were not, which is the divergence this file already calls out by name for
1670
+ // sessionsDir. One deployment value, one place, same rule as the profiles below it.
1671
+ resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, hostEnv: env, log }),
793
1672
  // #227. What PI_BACKENDS blessed, so a trigger naming an unblessed venue refuses pre-spend. The
794
1673
  // panel's picker is bounded by the same list, and this is the half that binds: the overlay is
795
1674
  // not the reviewed artifact (DES-PER-TRIGGER-SECRET-PROFILE).
@@ -807,10 +1686,14 @@ export async function startWorker(
807
1686
  neverStartedExits: backends.neverStartedExits,
808
1687
  prepareWorkspace: makePrepareWorkspace({
809
1688
  jobsDir: config.jobsDir,
1689
+ // Issue #464: the boot's own check, asked again before every job against the same env.
1690
+ ensureDir: (dir) => ensureJobsDirFn(dir, { env }),
810
1691
  forgeFor,
811
1692
  // REQ-RESURRECTABLE-SANDBOX: the deployment default, resolved per job against run.image so a
812
1693
  // retained directory records the image that actually ran and a sandbox re-opens that one.
813
1694
  jobImage: config.jobImage,
1695
+ // #277: the venue a retained directory records, which the sandbox refuses by when it is not here.
1696
+ defaultBackend: config.defaultBackend,
814
1697
  preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
815
1698
  // The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
816
1699
  // per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
@@ -823,24 +1706,7 @@ export async function startWorker(
823
1706
  // REQ-RESURRECTABLE-SANDBOX. With the window at 0 this IS the old bare `cleanup`, by the same
824
1707
  // `rm` on the same path -- a deployment that wants no retention keeps today's behaviour exactly.
825
1708
  cleanup: makeCleanup({ sandboxDir: config.sandboxDir, retentionHours: config.sandboxRetentionHours, log }),
826
- comment: async (job, text) => {
827
- // Best-effort: the processor awaits comment() inside its try, so a rejection here would
828
- // corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
829
- // This adapter NEVER throws.
830
- const forge = forgeFor(job);
831
- if (forge?.auth) {
832
- try {
833
- const token = await forge.auth.mintToken(job);
834
- await forge.host.postStatusComment(job, job.target, text, token);
835
- } catch (err) {
836
- log("comment_failed", { jobId: job?.id, reason: err?.message });
837
- }
838
- return;
839
- }
840
- // A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
841
- // post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
842
- log("comment", { jobId: job?.id, text });
843
- },
1709
+ comment,
844
1710
  log,
845
1711
  // Resolved per job so the credential always comes from the job's OWN forge. A job whose forge has
846
1712
  // no working auth refuses here, at mint time, rather than running anonymously -- and the refusal
@@ -852,7 +1718,22 @@ export async function startWorker(
852
1718
  // token comes from must always be something the trigger said.
853
1719
  mintToken: async (job) => {
854
1720
  const kind = job?.kind === "local" ? "github" : job?.kind;
855
- const auth = forges[kind]?.auth;
1721
+ // `ensureAuth` returns the boot-time auth when there is one, and otherwise re-resolves once
1722
+ // if boot's failure was transient (issue #316). It throws the transient failure through, and
1723
+ // that throw is UNTAGGED, so the arm below turns it into the retryable class rather than
1724
+ // letting it reach the processor's config classifier: "the forge was unreachable a moment
1725
+ // ago" is not "this deployment is misconfigured", and only one of those deserves a public
1726
+ // comment saying so.
1727
+ let auth;
1728
+ try {
1729
+ auth = await ensureAuth(kind);
1730
+ } catch (err) {
1731
+ // A determinate re-resolve failure carries the REAL reason (bad credentials, a key that
1732
+ // is not PKCS//8), which is strictly better than the generic message below, so it is
1733
+ // passed straight through rather than collapsed into it.
1734
+ if (err?.piDispatchConfig === true) throw err;
1735
+ throw new InfraRetry(`${kind} auth could not be resolved: ${err?.message ?? "unknown"}`);
1736
+ }
856
1737
  if (auth) return await auth.mintToken(job);
857
1738
  if (kind === "github") {
858
1739
  throw configError("github jobs and cron triggers with run.github require a working GITHUB_AUTH_SOURCE (gh/pat/app)");
@@ -867,104 +1748,350 @@ export async function startWorker(
867
1748
  },
868
1749
  });
869
1750
 
870
- // REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
871
- // where the operator is already looking. This is the local counterpart of the GitHub issue
872
- // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
873
- // what tells a human a run did nothing. The container's own output already streams via
874
- // runContainer's onOutput during the run.
875
- // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
876
- // job-image-missing), never
877
- // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
878
- // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
879
- // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
880
- // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
881
- const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
882
- for (const w of allWorkers) w.on("completed", (job, result) =>
883
- log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
884
- );
885
- for (const w of allWorkers) w.on("failed", (job, err) =>
886
- log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
887
- );
888
-
889
- // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
890
- // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
891
- // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
892
- const onStalled = makeStallGuard({
893
- redis,
894
- threshold: config.schedulerStallMax,
895
- // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
896
- // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
897
- removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
898
- log,
899
- });
900
- // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
901
- // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
902
- // so every stall threw a TypeError and the money backstop never counted one (issue #267).
903
- for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
904
-
905
- // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
906
- // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
907
- // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
908
- // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
909
- // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
910
- // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
911
- // right place to catch, because it can ask the registry and this cannot.
912
- const { served, unserved } = servedSchedules(schedules.current);
913
- for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
914
-
915
- if (served.length > 0) {
916
- // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
917
- // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
918
- // config" is correct again by construction and two hosts can no longer prune each other at all. The
919
- // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
920
- // disagreement, which no queue split can detect.
921
- const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
922
- try {
923
- const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
924
- log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
925
- } finally {
926
- await rq.close().catch(() => {});
1751
+ // FROM HERE TO THE RETURN, THE WORKER IS LIVE AND THE BOOT CAN STILL REFUSE (issue #299). The
1752
+ // Worker starts consuming at construction, so everything below runs beside a paid drain --
1753
+ // registrations, the stall guard, the boot reconcile (the reachable refusal, reproduced), the
1754
+ // watcher pushes, the retention sweep's construction and start. A refusal that merely threw left
1755
+ // the process printing an error while taking jobs, because `cli.mjs` sets `process.exitCode` and
1756
+ // the live Worker held the loop forever. The catch below covers the REGION, not a list of calls,
1757
+ // so a step added tomorrow is covered the day it is added.
1758
+ try {
1759
+
1760
+ // REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
1761
+ // where the operator is already looking. This is the local counterpart of the GitHub issue
1762
+ // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
1763
+ // what tells a human a run did nothing. The container's own output already streams via
1764
+ // runContainer's onOutput during the run.
1765
+ // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
1766
+ // provider-auth-refused | job-image-missing), never
1767
+ // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
1768
+ // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
1769
+ // BOTH workers, or a cron job on the host queue produces no `job_completed` line at all -- and
1770
+ // REQ-LOCAL-JOB-VISIBILITY's whole point is that a missing line is what tells a human a run did nothing.
1771
+ const allWorkers = [worker, ...(worker.hostWorker ? [worker.hostWorker] : [])];
1772
+ for (const w of allWorkers) w.on("completed", (job, result) => {
1773
+ log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) });
1774
+ // The hook's POLICY half (issue #288): worker-abort, runner-policy and provider-auth-refused RETURN, so they land here
1775
+ // and never in the failed listener -- a failed-only mount would miss exactly the paid terminals
1776
+ // the feature exists for. Folded into the existing listener body, never a second w.on: the
1777
+ // start-wiring harness records ONE handler per event, and two would race the log line's pin.
1778
+ if (onFailure && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason)) {
1779
+ onFailure({ jobId: job?.id, outcome: "policy", reason: result.reason });
1780
+ }
1781
+ });
1782
+ for (const w of allWorkers) w.on("failed", (job, err) => {
1783
+ log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) });
1784
+ // The one reason whose sentence is the fix itself (issue #458): logged WHOLE, beside the cut line above.
1785
+ if (err?.reason === NETNS_KEEPER_NOT_HOLDING || err?.reason === NETNS_KEEPER_CRASH_LOOP) log("job_failed_netns_keeper", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? "") });
1786
+ // The TERMINAL failed attempt only (issue #288): `finishedOn` is set by BullMQ's own move on the
1787
+ // non-retry branch alone, and the emit follows it, so this guard reads the queue's decision
1788
+ // instead of re-deriving attempts arithmetic that could drift from shouldRetry. A retried
1789
+ // attempt comments nothing (a flaky daemon must not post three comments for one recovery) and
1790
+ // pages nobody; a recovery comments nothing at all. This seam also catches what the processor
1791
+ // never sees: the stall-kill (maxStalledCount 0 fails a crashed worker's job at next pickup)
1792
+ // and the wait-gate rethrow that escapes above the processor's catch.
1793
+ if (job?.finishedOn) {
1794
+ void comment({ ...job.data, id: job.id }, (typeof err?.reason === "string" && Object.hasOwn(FAILED_COMMENT_BY_REASON, err.reason) ? FAILED_COMMENT_BY_REASON[err.reason] : null) ?? FAILED_COMMENT);
1795
+ onFailure?.({ jobId: job?.id, outcome: "failed", reason: typeof err?.reason === "string" ? err.reason : "infra" });
1796
+ }
1797
+ });
1798
+
1799
+ // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
1800
+ // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
1801
+ // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
1802
+ const onStalled = makeStallGuard({
1803
+ redis,
1804
+ threshold: config.schedulerStallMax,
1805
+ // The queue the schedulers were actually INSTALLED on. Torn down from `runtimeQueue` on a named
1806
+ // host, this money backstop -- a wedged scheduled run is re-paid on every stall -- silently no-ops.
1807
+ removeJobScheduler: (id) => cronQueue.removeJobScheduler(id),
1808
+ log,
1809
+ });
1810
+ // `makeStallGuard` returns the LISTENER, not an object holding one -- hence the name of the local. It was
1811
+ // called as `guard.onStalled(jobId)` here for the whole life of the feature, which is `undefined(jobId)`,
1812
+ // so every stall threw a TypeError and the money backstop never counted one (issue #267).
1813
+ for (const w of allWorkers) w.on("stalled", (jobId) => void onStalled(jobId));
1814
+
1815
+ // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
1816
+ // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
1817
+ // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
1818
+ // What this host will NOT be running, said once at boot and per trigger. A folder that belongs to
1819
+ // another machine is ordinary on a fleet; a folder that belongs to NO machine is a trigger that will
1820
+ // silently never fire, which is the silent no-op this project refuses -- and which `doctor` is the
1821
+ // right place to catch, because it can ask the registry and this cannot.
1822
+ const { served, unserved } = servedSchedules(schedules.current);
1823
+ for (const s of unserved) log("schedule_unserved", { schedulerId: s.schedulerId, reason: s.unserved });
1824
+
1825
+ if (served.length > 0) {
1826
+ // Onto the HOST queue when one is armed. That makes Gap 1 structural rather than merely gated: a
1827
+ // host queue's resident schedulers are only ever that host's, so `reconcile`'s "resident minus my
1828
+ // config" is correct again by construction and two hosts can no longer prune each other at all. The
1829
+ // fingerprint gate stays, because it still catches the divergence itself -- including a timezone
1830
+ // disagreement, which no queue split can detect.
1831
+ const rq = makeQueue(valkeyConn({ failFast: true }), { ...(hostQueue ? { name: hostQueue } : {}) });
1832
+ try {
1833
+ const r = await reconcileGated(rq, served, { registry, log, tz: hostTz, authored: authoredCron(config) });
1834
+ log("schedules_installed", { installed: r.installed, removed: r.removed, ...(unserved.length > 0 && { unserved: unserved.length }) });
1835
+ } finally {
1836
+ await rq.close().catch(() => {});
1837
+ }
1838
+ } else {
1839
+ log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
927
1840
  }
928
- } else {
929
- log("schedules_installed", { installed: 0, removed: 0, ...(unserved.length > 0 && { unserved: unserved.length }) });
930
- }
931
1841
 
932
- // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
933
- // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
934
- // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
935
- if (config.triggersFile) {
936
- watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
937
- }
1842
+ // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
1843
+ // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
1844
+ // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1845
+ // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
1846
+ if (config.triggersFile) {
1847
+ extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers));
1848
+ }
938
1849
 
939
- // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
940
- // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
941
- if (config.pauseWindowsFile) {
942
- watchPauseWindowsFile(config, pauseWindows, log);
943
- }
1850
+ // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
1851
+ // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
1852
+ if (config.pauseWindowsFile) {
1853
+ extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log, atBoot.pauseWindows));
1854
+ }
1855
+
1856
+ // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
1857
+ if (config.scopedLimitsFile) {
1858
+ extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log, atBoot.scopedLimits));
1859
+ }
944
1860
 
945
- // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
946
- if (config.scopedLimitsFile) {
947
- watchScopedLimitsFile(config, scopedLimits, log);
1861
+ // issue #292 / OQ-007: re-run the three retention sweeps on a timer, because the supported deployment
1862
+ // is a service that restarts only on failure, so the healthy worker was the one that never re-swept.
1863
+ // Armed HERE, at the end of boot beside the watches: all three closures exist, boot's own sweeps have
1864
+ // long finished, and the first tick lands one full interval later rather than during the schedules
1865
+ // reconcile, which is Redis-destructive work a small test interval would otherwise land inside.
1866
+ //
1867
+ // NOT CONSTRUCTED AT ALL when the knob is 0. That is how "0 is byte-identical to before" is a fact
1868
+ // rather than a claim: with no object there is no timer, no closer and no reachable second call to any
1869
+ // reaper. It is registered in `extraClosers` for issue #295's finding that UNREF'D IS NOT CLEANED UP --
1870
+ // this handle holds an `rmSync`, and a tick landing mid-drain could delete a retained workspace behind a
1871
+ // worker that already reported a clean shutdown.
1872
+ if (config.sweepIntervalHours > 0) {
1873
+ const sweep = makeRetentionSweepFn({
1874
+ reapers: [
1875
+ { name: "log", reap: reapLogs },
1876
+ { name: "sandbox", reap: reapSandboxes },
1877
+ { name: "session", reap: () => sessionStore.reapSessions() },
1878
+ ],
1879
+ intervalMs: config.sweepIntervalHours * 3600000,
1880
+ log,
1881
+ });
1882
+ sweep.start();
1883
+ extraClosers.push(sweep);
1884
+ }
1885
+
1886
+ log("worker_started", {
1887
+ queue: "pi-jobs",
1888
+ // Issue #278: the service's OWN answer to which daemon its jobs go to. `doctor` reads its caller's
1889
+ // shell, and a service's EnvironmentFile or a systemd User= can resolve differently.
1890
+ // Issue #354: all five null with `local` unblessed, where this host's docker CLI was not asked.
1891
+ dockerContext: bootEndpoint ? bootEndpoint.context : null,
1892
+ dockerEndpointLocal: bootEndpoint ? bootEndpoint.local : null,
1893
+ jobUser: bootDecision ? { mode: bootDecision.mode, user: bootDecision.user, cause: bootDecision.cause } : null, // issue #341
1894
+ // Issue #345: whether this daemon is observed applying a container's bounds and adding no mounts of its own; null = not read.
1895
+ daemonAppliesBounds: bootObserved ? bootObserved.observations.daemonAppliesBounds : null,
1896
+ runtimeAddsNoMounts: bootObserved ? bootObserved.observations.runtimeAddsNoMounts : null,
1897
+ // Issue #354: the podman venue's own boot answers, all null with `podman` unblessed (no `podman info` was asked):
1898
+ // the version it reported, whether it is rootless and whose uid its jobs run as, and its three observations.
1899
+ podmanVersion: bootPodmanRead?.answered ? bootPodmanRead.info.version : null,
1900
+ podmanRootless: bootPodmanRead?.answered ? bootPodmanRead.info.rootless : null,
1901
+ podmanJobUser: bootPodmanDecision ? { mode: bootPodmanDecision.mode, user: bootPodmanDecision.user, cause: bootPodmanDecision.cause, reason: bootPodmanDecision.reason } : null,
1902
+ podmanBoundsDelegated: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_BOUNDS_DELEGATED] : null,
1903
+ podmanAddsNoMounts: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_ADDS_NO_MOUNTS] : null,
1904
+ podmanServiceLocal: bootPodmanObserved ? bootPodmanObserved.observations[PODMAN_SERVICE_LOCAL] : null,
1905
+ host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
1906
+ imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
1907
+ concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
1908
+ sweepIntervalHours: config.sweepIntervalHours, // 0 = boot-only sweeps, this version's pre-#292 behaviour
1909
+ dailyCap: config.dailyCap,
1910
+ weeklyCap: config.weeklyCap, // null when the weekly window is disabled
1911
+ monthlyCap: config.monthlyCap, // null when the monthly window is disabled
1912
+ softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
1913
+ scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
1914
+ scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
1915
+ image: config.jobImage,
1916
+ valkey: config.valkeyUrl,
1917
+ // Issue #464: the literal address every Valkey client of this worker dials, beside the URL as written; null
1918
+ // where nothing was pinned (another machine's Valkey, dialled by name).
1919
+ valkeyPinned: valkey.pinned ? `${valkey.pinned.address.includes(":") ? `[${valkey.pinned.address}]` : valkey.pinned.address}:${valkey.pinned.port}` : null,
1920
+ logsDir: config.logsDir,
1921
+ settingsFile: config.settingsFile,
1922
+ captureJobLogs: config.captureJobLogs,
1923
+ logRetentionDays: config.logRetentionDays,
1924
+ sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
1925
+ });
1926
+ return worker;
1927
+ } catch (err) {
1928
+ // STOP WHAT WAS BUILT, THEN RETHROW, and both halves carry weight. The stop is the shutdown
1929
+ // minus the exit -- the cancel, the close, the closer drain, the client release -- so nothing is
1930
+ // left holding the loop and the process drains to the refusal's OWN exit code: entryExitCode
1931
+ // turns a tagged configError into EXIT_POLICY 2 and infra into the retryable 1, and a swallow
1932
+ // here would hand a supervisor a clean 0 for a boot that refused. `Promise.resolve`, not an
1933
+ // optional-chained `.catch`: a test double whose recording `stop` is synchronous would otherwise
1934
+ // raise a TypeError OVER the boot's real error, and the exit-code assertion meant to go red
1935
+ // under mutation would go green for the wrong reason. A synchronous throw from `stop` itself
1936
+ // still escapes, exactly as the closer loop documents for `extraClosers` -- a known bound.
1937
+ await Promise.resolve(worker?.stop?.()).catch(() => {});
1938
+ throw err;
948
1939
  }
1940
+ }
949
1941
 
950
- log("worker_started", {
951
- queue: "pi-jobs",
952
- host: config.workerName, // issue #57; `log` stamps it on every line, and the boot line names it where an operator looks first
953
- imageDigest: bootImage.imageDigest ?? null, // two hosts on two builds of one tag used to emit byte-identical boot lines
954
- concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
955
- dailyCap: config.dailyCap,
956
- weeklyCap: config.weeklyCap, // null when the weekly window is disabled
957
- monthlyCap: config.monthlyCap, // null when the monthly window is disabled
958
- softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
959
- scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
960
- scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
961
- image: config.jobImage,
962
- valkey: config.valkeyUrl,
963
- logsDir: config.logsDir,
964
- settingsFile: config.settingsFile,
965
- captureJobLogs: config.captureJobLogs,
966
- logRetentionDays: config.logRetentionDays,
967
- sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
1942
+ /**
1943
+ * The boot refusal text for a job-user decision, or `null` to boot. Exported so its default-venue branch is pinned on the
1944
+ * predicate as well as end to end (a podman default reaches it since issue #354 part 2, and must not refuse on local's causes).
1945
+ */
1946
+ export function jobUserBootRefusal(decision, defaultBackend) {
1947
+ if (decision?.mode !== "unmappable" || !BOOT_REFUSING_JOB_USER_CAUSES.has(decision.cause)) return null;
1948
+ return defaultBackend === DEFAULT_BACKEND ? jobUserRefusal(decision) : null;
1949
+ }
1950
+
1951
+ /**
1952
+ * The boot refusal text for a podman job-user decision, or `null` to boot (issue #354): an identity cause no podman job
1953
+ * can get past (`PODMAN_BOOT_REFUSING_CAUSES`), and only while `podman` is the DEFAULT venue, `jobUserBootRefusal`'s rule.
1954
+ * Kept apart from that function rather than merged into it, because the two venues' causes share a name
1955
+ * (`worker-is-root`) and not a remedy, and each venue's text comes from its own map.
1956
+ */
1957
+ export function podmanBootRefusal(decision, defaultBackend) {
1958
+ if (defaultBackend !== PODMAN_BACKEND) return null;
1959
+ if (decision?.mode !== "unmappable" || !PODMAN_BOOT_REFUSING_CAUSES.has(decision.cause)) return null;
1960
+ return podmanJobUserRefusal(decision);
1961
+ }
1962
+
1963
+ /**
1964
+ * The boot refusal for a rootful Podman containers.conf that reaches a local job (issue #448), from `observeRootfulConf`'s
1965
+ * answer, or `null`: only while `local` is the default venue, and not when the local job user is already refused (that
1966
+ * refusal, earlier at boot or per job, is the one to fix first). `transient` (thrown untagged, exit 1, so the supervisor
1967
+ * restarts it) is a read that failed for a moment AND, since gate round 1 of PR #473, a service older than its
1968
+ * containers.conf or a change time ahead of the clock: each heals by itself, so none may strand a unit with
1969
+ * `RestartPreventExitStatus=2`. Exported for the doc test, as `podmanConfBootRefusal` is.
1970
+ */
1971
+ export function localConfBootRefusal(decision, defaultBackend, rootful) {
1972
+ if (defaultBackend !== DEFAULT_BACKEND || !rootful?.refusal || decision?.mode === "unmappable") return null;
1973
+ return { message: rootfulConfRefusal(rootful.refusal), transient: rootfulConfRetries(rootful.refusal) };
1974
+ }
1975
+
1976
+ /**
1977
+ * A rootful finding as the processor's `podmanConfRefused` (issue #448): the podman venue's shape, marked `rootful`.
1978
+ * `retry` marks what the processor throws for the queue's retry rather than returns (a moment's read failure, a service
1979
+ * older than its containers.conf, a clock behind a change time), with the evidence its retry message names.
1980
+ */
1981
+ function rootfulRefused(found) {
1982
+ return {
1983
+ reason: found.cause,
1984
+ key: found.key ?? null,
1985
+ message: rootfulConfRefusal(found),
1986
+ rootful: true,
1987
+ ...(found.restart ? { restart: true } : {}),
1988
+ ...(found.skew ? { skew: true } : {}),
1989
+ ...(found.transient ? { transient: true } : {}),
1990
+ ...(rootfulConfRetries(found) ? { retry: true, evidence: found.evidence } : {}),
1991
+ };
1992
+ }
1993
+
1994
+ /**
1995
+ * The boot refusal for a widening containers.conf on the podman venue (issue #428), as `{ message, transient }`, or
1996
+ * `null` to boot: only while `podman` is the DEFAULT venue, `podmanBootRefusal`'s rule, and not when the identity is
1997
+ * already refused (that refusal names the fix that comes first, and with `podman` merely blessed its jobs are refused
1998
+ * one by one anyway). `transient` is a read that failed for a moment, which the caller throws untagged.
1999
+ */
2000
+ export function podmanConfBootRefusal(decision, defaultBackend, files) {
2001
+ if (defaultBackend !== PODMAN_BACKEND || !decision || decision.mode === "unmappable") return null;
2002
+ const widened = podmanConfWidening(files);
2003
+ return widened ? { message: podmanConfRefusal(widened), transient: widened.transient === true } : null;
2004
+ }
2005
+
2006
+ /** What makes two job-user decisions the same for the `job_user` log line. */
2007
+ function jobUserLogKey(decision) {
2008
+ return `${decision?.mode}|${decision?.user ?? ""}|${decision?.cause ?? ""}|${decision?.reason ?? ""}`;
2009
+ }
2010
+
2011
+ /** A one-line summary of an endpoint answer, used only to notice that it changed. */
2012
+ function dockerEndpointState(endpoint) {
2013
+ return `${endpoint.local}|${endpoint.context ?? ""}|${endpoint.endpoint ?? ""}|${endpoint.reason ?? ""}`;
2014
+ }
2015
+
2016
+ /**
2017
+ * What an endpoint answer shows, for a refusal message. The context name is operator config; the endpoint is
2018
+ * the display form, which is the value reduced to `scheme://host` or withheld, never edited (`displayEndpoint`).
2019
+ */
2020
+ function dockerEndpointEvidence(endpoint) {
2021
+ if (endpoint.local === null) return `the docker CLI did not say which endpoint it resolves (${endpoint.reason})`;
2022
+ return `the docker CLI resolves context ${quotedShown(endpoint.context)} to ${endpointShown(endpoint)}${endpoint.local ? ", on this host" : ", which is not shown to be on this host"}`;
2023
+ }
2024
+
2025
+ /** Log an endpoint answer that is not plainly local; a return to local is logged only as a change. */
2026
+ function logDockerEndpoint(log, endpoint, { changed = false } = {}) {
2027
+ if (endpoint.local === false) log("docker_endpoint_not_local", { context: endpoint.context, endpoint: endpoint.endpoint });
2028
+ else if (endpoint.local === null) log("docker_endpoint_unresolved", { reason: endpoint.reason });
2029
+ else if (changed) log("docker_endpoint_local", { context: endpoint.context, endpoint: endpoint.endpoint });
2030
+ }
2031
+
2032
+ /**
2033
+ * Where the worker's clients stand (issue #464): enforced exactly as its boot judgement was (`valkey.enforce`), so a
2034
+ * reconnect is judged by the rule the boot applied, never a looser one; PI_VALKEY_SHARED from the deployment `.env`.
2035
+ */
2036
+ export function workerValkeyContext(valkey, env, { cwd = process.cwd(), readEnv } = {}) {
2037
+ return valkeyClientContext({ env, cwd, rootRefused: valkey?.rootRefused === true, ...(readEnv ? { readEnv } : {}) });
2038
+ }
2039
+
2040
+ /**
2041
+ * The worker's real Valkey judgement (issue #464, gate round 2): `resolveWorkerValkey` with this host's facts. The
2042
+ * deployment `.env` is the one in the working directory (the unit's WorkingDirectory is the deployment folder), read
2043
+ * only for PI_VALKEY_SHARED; nothing else of it is loaded into this process.
2044
+ */
2045
+ async function defaultJudgeValkey({ url, venues, env }) {
2046
+ const envPath = join(process.cwd(), ".env");
2047
+ let envText = null;
2048
+ try {
2049
+ // Bytes (gate round 3): the hardened reader checks what systemd refuses to load before it decodes.
2050
+ envText = readFileSync(envPath);
2051
+ } catch (err) {
2052
+ // No .env here: nothing opts in. Any other failure is said, never read as "no opt-in" (gate round 3).
2053
+ if (err?.code !== "ENOENT") throw configError(`${envPath} could not be read (${err?.code ?? err?.message}), and the worker reads PI_VALKEY_SHARED from it, so it does not start`);
2054
+ }
2055
+ const fs = { readFileSync };
2056
+ const euid = process.geteuid?.();
2057
+ let user = null;
2058
+ try {
2059
+ user = userInfo().username;
2060
+ } catch {
2061
+ // A uid with no passwd entry: its subordinate ranges are read by uid.
2062
+ }
2063
+ const valkey = await resolveWorkerValkey({
2064
+ url,
2065
+ venues,
2066
+ platform: process.platform,
2067
+ env,
2068
+ envText,
2069
+ envPath,
2070
+ probeTcp: probeTcpAddress,
2071
+ lookup: (host, opts) => dnsLookup(host, opts),
2072
+ fs,
2073
+ euid,
2074
+ user,
2075
+ ownerName: (uid) => passwdNameFrom(fs, uid),
2076
+ interfaces: networkInterfaces,
2077
+ subuids: readSubuidRanges({ user, euid, fs }),
2078
+ configError,
968
2079
  });
969
- return worker;
2080
+ await refuseValkeyAuth(valkey, env);
2081
+ return valkey;
2082
+ }
2083
+
2084
+ /**
2085
+ * Issue #468: the worker's credential asked once at boot, on the judged address, before anything is built on it. A
2086
+ * Valkey that requires a password this worker does not send (NOAUTH), or refuses the one it sends (WRONGPASS), is a
2087
+ * configError (exit 2, not restarted into the same answer), naming VALKEY_PASSWORD and never its value: left to the
2088
+ * clients, it was an endless stream of NOAUTH errors from a worker that looked alive. Nothing answering is left to the
2089
+ * clients' own retries, as before. Exported for its test; `authState` is the seam.
2090
+ */
2091
+ export async function refuseValkeyAuth(valkey, env, { cwd = process.cwd(), authState = valkeyAuthState } = {}) {
2092
+ const context = workerValkeyContext(valkey, env, { cwd });
2093
+ const { state, error } = await authState(valkey.url, { context, servername: valkey.servername ?? null });
2094
+ // Gate round 2 of PR #478: a database the server does not have (`/16` on a default Valkey) is a refusal, exit 2.
2095
+ if (state === "dbrange") throw configError(error);
2096
+ if (state === "noauth" || state === "wrongpass") throw configError(authRefusalFor(state, valkey.url, context));
970
2097
  }