@edgehero/pi-dispatch 1.10.3 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.example +303 -150
  2. package/README.md +52 -0
  3. package/deploy/com.pi-dispatch.worker.plist +10 -4
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +12 -1
  15. package/deploy/worker-env-wrapper.sh +63 -37
  16. package/deploy/worker.service +18 -8
  17. package/package.json +15 -5
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4756 -394
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +456 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +245 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-pi.mjs +19 -3
  54. package/src/host-registry.mjs +29 -2
  55. package/src/identity.mjs +29 -4
  56. package/src/image-preflight.mjs +46 -11
  57. package/src/image-ref.mjs +21 -0
  58. package/src/index.mjs +363 -13
  59. package/src/init.mjs +197 -38
  60. package/src/job-user.mjs +252 -0
  61. package/src/json-duplicates.mjs +204 -0
  62. package/src/live-probes.mjs +1020 -0
  63. package/src/materialize.mjs +4 -11
  64. package/src/netns-keeper.mjs +264 -0
  65. package/src/on-failure.mjs +119 -0
  66. package/src/outbox.mjs +7 -0
  67. package/src/packages.mjs +2 -2
  68. package/src/podman-stack.mjs +1304 -0
  69. package/src/prepare-github.mjs +6 -6
  70. package/src/prepare-local.mjs +51 -17
  71. package/src/prepare.mjs +27 -6
  72. package/src/pricing.mjs +9 -5
  73. package/src/processor.mjs +506 -26
  74. package/src/provider-key.mjs +66 -0
  75. package/src/provider-steering.mjs +185 -0
  76. package/src/queue.mjs +35 -8
  77. package/src/redact.mjs +84 -0
  78. package/src/reserved-env.mjs +7 -3
  79. package/src/retention-sweep.mjs +178 -0
  80. package/src/run-container.mjs +181 -14
  81. package/src/run-history.mjs +105 -16
  82. package/src/runtime-observations.mjs +1152 -0
  83. package/src/runtime-settings.mjs +13 -8
  84. package/src/sandbox-cli.mjs +100 -95
  85. package/src/sandbox-store.mjs +612 -45
  86. package/src/sandbox.mjs +1459 -37
  87. package/src/schedules.mjs +16 -3
  88. package/src/secret-profiles.mjs +2 -1
  89. package/src/secrets.mjs +24 -6
  90. package/src/service-env.mjs +247 -0
  91. package/src/service.mjs +618 -28
  92. package/src/session-store.mjs +678 -53
  93. package/src/start.mjs +1348 -326
  94. package/src/subscriptions.mjs +7 -3
  95. package/src/transient.mjs +240 -0
  96. package/src/triggers-file.mjs +71 -15
  97. package/src/triggers.mjs +179 -19
  98. package/src/up.mjs +1399 -85
  99. package/src/valkey-auth.mjs +529 -0
  100. package/src/valkey-endpoint.mjs +367 -0
  101. package/src/watch-closer.mjs +158 -0
package/src/backends.mjs CHANGED
@@ -2,9 +2,10 @@
2
2
  * THE BACKEND TABLE -- the one place that says where a job's container can run, and what each place is
3
3
  * ABLE to guarantee (issue #227).
4
4
  *
5
- * A backend is where the box is built. Today there is one, `local`, which is the Docker daemon on the
6
- * worker's own host and is what every deployment has always used. The table exists so a second one can be
7
- * added without reading the worker's source, and so that adding one cannot quietly weaken a control.
5
+ * A backend is where the box is built. `local` is the Docker daemon on the worker's own host and is what
6
+ * every deployment has always used; since issue #354 `podman` is the worker account's own rootless Podman.
7
+ * The table exists so another can be added without reading the worker's source, and so that adding one
8
+ * cannot quietly weaken a control.
8
9
  *
9
10
  * This module imports NOTHING, deliberately, for `forges.mjs`'s reason: it is a leaf, and `doctor`, the
10
11
  * config loader and (later) the receiver all need to read a declaration without pulling the Docker
@@ -62,10 +63,12 @@
62
63
  * and the same is true one level out. The value of the table is not that a vendor is verified; it is that a
63
64
  * MISMATCH becomes a refusal instead of a silent downgrade. `backend-conformance.mjs` verifies THREE of the
64
65
  * thirteen -- exit-code fidelity, the abort flag's independence from the code, and the read-only downgrade a
65
- * copying runtime takes -- plus the shape of a bundle and the internal consistency of its declaration. The
66
- * other ten need a live container on the target runtime, and the harness names each one and what it would
67
- * take rather than passing them in silence. So a backend can still declare all of this and do most of it:
68
- * these words are a contract with three of them checked, which is more than none and less than proof.
66
+ * copying runtime takes -- plus the shape of a bundle and the internal consistency of its declaration. EIGHT more
67
+ * are read back off a live container (issues #278 and #344): for `local` by `doctor --live` (`live-probes.mjs`), for
68
+ * any other runtime through the harness's `readBack` probe, which the adapter supplies. The remaining TWO the harness
69
+ * names with what each would take rather than passing them in silence. So a backend can still declare all of this
70
+ * and do most of it: these words are a contract with three of them checked and eight readable, which is more than
71
+ * none and less than proof -- the read-back is of a fixture with the job image, not of every job.
69
72
  *
70
73
  * THE PROPERTIES ARE NOT A RENDERING OF `ISOLATION_FLAGS`, and must not become one. That array is the
71
74
  * literal, value-free, unconditional set the local argv splices in, and two tests assert every member of it
@@ -78,8 +81,10 @@
78
81
  /**
79
82
  * The exit codes a DOCKER-shaped runtime uses for "the runner never ran": 125 is `docker run` itself
80
83
  * failing, 126 an entrypoint that is not executable, 127 an entrypoint that was not found. In all three the
81
- * daemon never handed control to the runner, so nothing was spent -- which is what `container-never-started`
82
- * means, and why they refund the budget slot instead of keeping it.
84
+ * daemon normally never handed control to the runner, so nothing was spent -- which is what `container-never-started`
85
+ * means, and why they refund the budget slot instead of keeping it. NOT always (issue #345, measured): a CLI that lost
86
+ * the daemon's API mid-run exits 125 while its container runs on, which `run-container.mjs` finds through the cidfile
87
+ * and reports as `detached`, unrefunded.
83
88
  *
84
89
  * HERE, in the leaf, rather than in `backend-local.mjs`, so `processor.mjs` can default to them without
85
90
  * importing the local adapter and dragging `node:child_process` and the docker CLI machinery into its
@@ -263,14 +268,301 @@ export function declarationOf(name, property) {
263
268
  // Only meaningful for an asserted word, and null otherwise rather than absent, so a consumer that
264
269
  // prints it unconditionally renders nothing rather than "undefined".
265
270
  assertedBy: word === ASSERTED ? (entry.asserts?.[property] ?? null) : null,
271
+ // The observation this word holds only while (#278), or null. A consumer that prints the word without it
272
+ // would print `enforced` for a daemon that is elsewhere.
273
+ observedBy: entry.observedBy?.[property] ?? null,
266
274
  };
267
275
  }
268
276
 
277
+ /**
278
+ * The word a backend EARNS for a property given what is observed on this host (issue #278). The declared word
279
+ * when the property is not observation-gated or its observation is `true`; otherwise at most `asserted`, since
280
+ * the guarantee then rests on whoever arranged for the observation to be false. `undefined` observations get no
281
+ * credit, this file's standing polarity. `undefined` for an unknown backend or property.
282
+ */
283
+ export function effectiveWord(name, property, observations = {}) {
284
+ const entry = backendFor(name);
285
+ if (!entry || !isProperty(property)) return undefined;
286
+ const word = entry.declares[property] ?? ABSENT;
287
+ const observation = entry.observedBy?.[property];
288
+ if (!observation || observations?.[observation] === true) return word;
289
+ return (RANK[word] ?? 0) > RANK[ASSERTED] ? ASSERTED : word;
290
+ }
291
+
292
+ /**
293
+ * Which floored properties a backend declares strongly enough but does not EARN given the observations, as
294
+ * `[{ backend, property, have, want, observedBy }]` -- the floor misses the observation alone causes. The
295
+ * capability misses stay `floorShortfall`'s, checked at config load; this is the half only a live read can
296
+ * answer, so it is checked where the read happens (boot, and before each job's spend).
297
+ */
298
+ export function unobservedFloor(backends, floor, observations = {}) {
299
+ const out = [];
300
+ for (const backend of backends ?? []) {
301
+ const entry = backendFor(backend);
302
+ if (!entry) continue;
303
+ for (const [property, want] of Object.entries(floor ?? {})) {
304
+ if (!isProperty(property) || !isDeclaration(want) || want === ABSENT) continue;
305
+ const declared = entry.declares[property] ?? ABSENT;
306
+ const have = effectiveWord(backend, property, observations);
307
+ if (meets(declared, want) && !meets(have, want)) out.push({ backend, property, have, want, observedBy: entry.observedBy[property] });
308
+ }
309
+ }
310
+ return out;
311
+ }
312
+
313
+ /**
314
+ * The refusals `unobservedFloor` implies, as operator-facing messages; empty when the floor holds. Outside
315
+ * `backendRefusals` because `loadConfig` is synchronous and cannot observe anything; the caller that made the
316
+ * observation passes it, with `evidence` -- `{ [observation]: "what was seen" }` -- for the message.
317
+ */
318
+ export function observationRefusals({ backends = [], backendFloor = {}, observations = {}, evidence = {}, only = null } = {}) {
319
+ // `only` narrows the check to the named observations, for a caller that has read just those so far (boot reads the
320
+ // endpoint before the daemon, and must not refuse on an observation it has not asked yet).
321
+ const misses = unobservedFloor(backends, backendFloor, observations).filter((m) => !only || only.includes(m.observedBy));
322
+ if (misses.length === 0) return [];
323
+ const lines = misses.map((m) => ` ${m.backend}: ${m.property}=${m.want} holds only while ${OBSERVATIONS[m.observedBy] ?? m.observedBy}; ${evidence[m.observedBy] ?? "that was not observed"}`);
324
+ // One remedy per observation that missed, in the closed list's order, so a bounds miss is never told to repoint the CLI.
325
+ const remedies = Object.keys(OBSERVATION_FIX).filter((o) => misses.some((m) => m.observedBy === o)).map((o) => OBSERVATION_FIX[o]);
326
+ return [`PI_BACKEND_FLOOR asks for something this host is not observed to provide:\n${lines.join("\n")}\n${remedies.join("\n")}`];
327
+ }
328
+
329
+ /**
330
+ * Whether the floor misses `observationRefusals` would report rest ONLY on observations that were not answered (`null`
331
+ * or `undefined`, as opposed to an answered `false`): a daemon still starting, a read that timed out. Such a refusal is
332
+ * transient, and the caller retries rather than refusing for good. `false` when nothing misses.
333
+ */
334
+ export function observationRefusalIsTransient({ backends = [], backendFloor = {}, observations = {} } = {}) {
335
+ const missed = unobservedFloor(backends, backendFloor, observations).map((m) => m.observedBy);
336
+ return missed.length > 0 && missed.every((o) => typeof observations?.[o] !== "boolean");
337
+ }
338
+
269
339
  /** Is `name` one of the closed list? `Object.hasOwn`, so `"toString"` is not a property. */
270
340
  export function isProperty(name) {
271
341
  return Object.hasOwn(PROPERTIES, name);
272
342
  }
273
343
 
344
+ /** The observation that the docker CLI on this host resolves an endpoint on this host (issue #278). */
345
+ export const DOCKER_ENDPOINT_LOCAL = "dockerEndpointLocal";
346
+
347
+ /**
348
+ * The observation that the daemon reports it applies a container's pid and memory bounds (issue #345), from the same
349
+ * `docker info` read the job user is decided from. `true` only for `PidsLimit` and `MemoryLimit` both true on a daemon
350
+ * that is neither rootless nor Podman: Podman's Docker API hard-codes `PidsLimit` and derives `MemoryLimit` from the root
351
+ * cgroup, so on Podman the booleans say nothing about a container (source; measured true on a rootless host applying
352
+ * neither). CPU is never read: rootful Podman reports `CpuCfsQuota: false` while applying `--cpus` (measured).
353
+ */
354
+ export const DAEMON_APPLIES_BOUNDS = "daemonAppliesBounds";
355
+
356
+ /**
357
+ * The observation that the container runtime mounts nothing of its own into a job container (issue #345). Docker adds
358
+ * none; rootful Podman mounts `/run/secrets` from its default `mounts.conf` into every container, invisible in
359
+ * `.Mounts` (measured: host subscription files readable by the job user). On Podman it holds only with the documented
360
+ * empty `/etc/containers/mounts.conf` override and no `volumes`/`mounts` key in any containers.conf this host reads
361
+ * (`runtime-observations.mjs`).
362
+ */
363
+ export const RUNTIME_ADDS_NO_MOUNTS = "runtimeAddsNoMounts";
364
+
365
+ /**
366
+ * The observation that a ROOTLESS Podman applies a job container's pid, memory and cpu bounds (issue #354; the facts it
367
+ * rests on corrected by issue #453's measurements). From the `podman info` read the venue's job user is decided from
368
+ * (rootless, cgroup v2, the cgroup manager Podman uses) and this host's files: the account's systemd user manager
369
+ * running with `pids`, `memory` and `cpu` delegated to it (its `user@<uid>.service` cgroup), and Podman able to put a
370
+ * container under it (the `systemd` cgroup manager, or the worker itself inside that manager). Measured on Podman
371
+ * 5.8.1, Fedora 44: without a reachable user manager a container lands in the caller's own cgroup and its bounds read
372
+ * back `max`, exit 0, and `--cgroups=disabled` does the same; so the word rests on those facts being OBSERVED, and a
373
+ * `cgroups` key in a containers.conf this user's Podman reads withholds it. `pi-dispatch doctor --live` reads
374
+ * `pids.max` and `memory.max` back off a container.
375
+ */
376
+ export const PODMAN_BOUNDS_DELEGATED = "podmanBoundsDelegated";
377
+
378
+ /**
379
+ * The observation that this worker account's rootless Podman adds no mounts of its own to a job container (issue #354):
380
+ * the mounts.conf that wins (the user's `~/.config/containers/mounts.conf` when it exists, which OVERRIDES
381
+ * `/etc/containers/mounts.conf`, measured) is empty, no containers.conf this user's Podman reads sets `volumes`,
382
+ * `mounts`, `devices` or `hooks_dir`, no OCI hook is installed and FIPS mode is off (`runtime-observations.mjs`,
383
+ * `observePodman`).
384
+ */
385
+ export const PODMAN_ADDS_NO_MOUNTS = "podmanAddsNoMounts";
386
+
387
+ /**
388
+ * The observation that the podman CLI this worker spawns runs its containers in-process on this host rather than through
389
+ * a remote service (issue #354): `podman info` reports `host.serviceIsRemote: false`. `CONTAINER_HOST`, `--remote` or a
390
+ * `containers.conf` service destination turn it `true` (measured), and a remote service is where the provider key and
391
+ * the per-job forge token would travel as `-e NAME=VALUE`.
392
+ */
393
+ export const PODMAN_SERVICE_LOCAL = "podmanServiceLocal";
394
+
395
+ /**
396
+ * The podman venue's refusal of an account containers.conf that widens every job (issue #428), and the run record's
397
+ * `reason` for a job it refuses. Here, in the leaf, rather than in backend-podman.mjs beside the check, so the processor
398
+ * can name it without importing the venue's module (which pulls in the run path and pi-ai).
399
+ */
400
+ export const PODMAN_CONF_WIDENS_JOB = "podman-conf-widens-job";
401
+
402
+ /**
403
+ * The containers.conf keys that refusal refuses on presence (issues #428, #450 and #448), in the order every text names
404
+ * them. The ONE list for the podman venue: backend-podman.mjs builds `WIDENING_KEY` and its refusal texts from it, and
405
+ * `podman-doc.test.mjs` requires every list of these keys in the specs, the docs and the source to be exactly this one,
406
+ * since three copies drifted when #450 added two. The rule for a key issue #448 measured: REFUSED when the job saw its
407
+ * value cross a boundary the worker's argv sets (what the job may reach, run, read or be limited by), or when it could
408
+ * not be measured; documented as INERT when the argv's own pins overrode it; documented as HARMLESS
409
+ * (`PODMAN_HARMLESS_KEYS`) when it reached the job but moved nothing that is a boundary. Measured with the podman venue's
410
+ * own argv on rootless Podman 5.8.1 (Fedora 44) and 4.9.3 (Ubuntu 24.04), each key alone in the account's own
411
+ * containers.conf, on `--network=private` and on an `--internal` network, one parenthesis each:
412
+ * `default_sysctls` (a sysctl set in the job, the vendor's own block excepted), `default_ulimits` (a ulimit set),
413
+ * `seccomp_profile` (the job ran under the named profile), `init_path` (the named binary as its PID 1), `dns_servers`
414
+ * (its nameserver, on `--network=private`), `dns_options` (its resolver options), `dns_searches` (its search list),
415
+ * `base_hosts_file` (its /etc/hosts), `oom_score_adj` (its OOM score), `privileged` (a full capability bounding set, no
416
+ * seccomp filter, the host's devices and an unconfined SELinux label, though `--cap-drop=ALL` still empties its
417
+ * effective set), and from gate round 1 of PR #473: `label` (false ran it as `spc_t` on Fedora), `cgroup_conf` (its
418
+ * pids.max became max past `--pids-limit`), `host_containers_internal_ip` (the address it reaches as
419
+ * host.containers.internal), `runtimes` (the `[engine.runtimes]` table: a wrapper runtime ran for every job),
420
+ * `conmon_path` (a wrapper conmon ran for every job), `cgroups` (disabled left its pids and memory bounds unapplied) and
421
+ * `umask` (its umask became the one named, so what it writes to the host is as open as that says).
422
+ */
423
+ export const PODMAN_WIDENING_KEYS = Object.freeze([
424
+ "pasta_options",
425
+ "network_cmd_options",
426
+ "annotations",
427
+ "env",
428
+ "helper_binaries_dir",
429
+ "network_cmd_path",
430
+ "default_sysctls",
431
+ "default_ulimits",
432
+ "seccomp_profile",
433
+ "init_path",
434
+ "dns_servers",
435
+ "dns_options",
436
+ "dns_searches",
437
+ "base_hosts_file",
438
+ "oom_score_adj",
439
+ "privileged",
440
+ "label",
441
+ "cgroup_conf",
442
+ "host_containers_internal_ip",
443
+ "runtimes",
444
+ "conmon_path",
445
+ "cgroups",
446
+ "umask",
447
+ ]);
448
+
449
+ /**
450
+ * The keys of `PODMAN_WIDENING_KEYS` that shape this account's shared rootless network helper (pasta or slirp4netns, the
451
+ * program behind it, and `env`, whose `[engine]` form can point Podman at another containers.conf that does): only these
452
+ * leave a running network carrying the setting after the key is gone, so only their remedy resets that network (issue
453
+ * #450). Every other refused key is applied per container, and the next job after its removal runs (gate round 1 of PR
454
+ * #473, which found the reset asked for every key).
455
+ */
456
+ export const PODMAN_NETWORK_HELPER_KEYS = Object.freeze(["pasta_options", "network_cmd_options", "env", "helper_binaries_dir", "network_cmd_path"]);
457
+
458
+ /**
459
+ * The keys MEASURED INERT for a podman venue job (issue #448), on rootless Podman 5.8.1 and 4.9.3 with the venue's own
460
+ * argv, one parenthesis each: `userns` (the argv's `--userns=keep-id`), `pidns` (its `--pid=private`), `ipcns` (its
461
+ * `--ipc=private`), `utsns` (its `--uts=private`), `cgroupns` (its `--cgroupns=private`), `netns` (its `--network`),
462
+ * `apparmor_profile` (rootless Podman applies no AppArmor profile: the job's label was `crun (unconfined)` with and
463
+ * without it on Ubuntu 24.04), `default_capabilities` (`--cap-drop=ALL`), `no_new_privileges` (`no-new-privileges`),
464
+ * `init` (`--init`), `pids_limit` (`--pids-limit`), `shm_size` (`--shm-size`), `env_host` (`--env-host=false`) and
465
+ * `http_proxy` (`--http-proxy=false`: a proxy in the account's environment did not reach the job). Documented, not
466
+ * refused.
467
+ */
468
+ export const PODMAN_ROOTLESS_INERT_KEYS = Object.freeze(["userns", "pidns", "ipcns", "utsns", "cgroupns", "netns", "apparmor_profile", "default_capabilities", "no_new_privileges", "init", "pids_limit", "shm_size", "env_host", "http_proxy"]);
469
+
470
+ /**
471
+ * The containers.conf keys the `local` venue refuses where rootful Podman's Docker API service on this host runs its jobs
472
+ * (issue #448), whatever the value, under `PODMAN_WIDENING_KEYS`' rule, each MEASURED with the job argv the worker
473
+ * builds, on the default network and on an `--internal` one, on Fedora 44 with rootful Podman 5.8.1 and on Ubuntu 24.04
474
+ * with 4.9.3 (`apparmor_profile` on Ubuntu alone, which runs AppArmor; `label` reached the job on Fedora alone, which runs
475
+ * SELinux). What each did, key by key, one parenthesis each so no two keys are read as a list:
476
+ * `annotations` (the job got the service's own supplementary groups), `env` (a variable in every job),
477
+ * `helper_binaries_dir` (the netavark and aardvark-dns Podman ran as root for the job's network were the named
478
+ * directory's), `default_sysctls` (a sysctl set in the job), `default_ulimits` (a ulimit set in the job), `userns`
479
+ * (with `auto` and no subordinate range the container could not be created), `pidns` (a host PID namespace, refused
480
+ * at create against the argv's `--init`), `ipcns` (a host IPC namespace, refused at create against `--shm-size`),
481
+ * `utsns` (the host's UTS namespace and hostname), `cgroupns` (the host's cgroup namespace), `netns` (the host's
482
+ * network namespace, on the default network), `seccomp_profile` (the job ran under the named profile),
483
+ * `apparmor_profile` (`unconfined` replaced the job's `containers-default` profile), `init_path` (the job's PID 1 was
484
+ * the named binary), `dns_servers` (its resolv.conf nameserver, default network), `dns_options` (its resolv.conf
485
+ * options), `dns_searches` (its resolv.conf search list), `base_hosts_file` (its /etc/hosts), `label` (false ran it as
486
+ * `spc_t`), `cgroup_conf` (its pids.max became max), `host_containers_internal_ip` (the address of
487
+ * host.containers.internal), `runtimes` (a wrapper runtime ran for every job), `conmon_path` (a wrapper conmon ran for
488
+ * every job) and `cgroups` (disabled left its pids and memory bounds unapplied).
489
+ * The one value that is not refused is the vendor's own `default_sysctls = ["net.ipv4.ping_group_range=0 0"]`,
490
+ * uncommented in the stock containers.conf of Fedora 44 and Ubuntu 24.04 (`STOCK_CONF_BLOCKS` in
491
+ * runtime-observations.mjs). The rest are `PODMAN_ROOTFUL_INERT_KEYS` and `PODMAN_HARMLESS_KEYS`. The first three are in
492
+ * `PODMAN_WIDENING_KEYS`' order, and `podman-doc.test.mjs` holds every three-key list of these keys to exactly one of the
493
+ * derived lists.
494
+ */
495
+ export const PODMAN_ROOTFUL_WIDENING_KEYS = Object.freeze([
496
+ "annotations",
497
+ "env",
498
+ "helper_binaries_dir",
499
+ "default_sysctls",
500
+ "default_ulimits",
501
+ "userns",
502
+ "pidns",
503
+ "ipcns",
504
+ "utsns",
505
+ "cgroupns",
506
+ "netns",
507
+ "seccomp_profile",
508
+ "apparmor_profile",
509
+ "init_path",
510
+ "dns_servers",
511
+ "dns_options",
512
+ "dns_searches",
513
+ "base_hosts_file",
514
+ "label",
515
+ "cgroup_conf",
516
+ "host_containers_internal_ip",
517
+ "runtimes",
518
+ "conmon_path",
519
+ "cgroups",
520
+ ]);
521
+
522
+ /**
523
+ * The keys MEASURED INERT for a rootful local job (issue #448), on both hosts, one parenthesis each: `pasta_options`
524
+ * (rootful Podman runs no pasta for these networks; netavark sets them up), `network_cmd_options` (nor slirp4netns),
525
+ * `network_cmd_path` (no slirp4netns ran), `default_capabilities` (the argv's `--cap-drop=ALL`), `no_new_privileges`
526
+ * (the argv's `no-new-privileges`), `init` (the argv's `--init`), `oom_score_adj` (the compat API sends its own, on
527
+ * Fedora and on Ubuntu), `pids_limit` (the argv's `--pids-limit`), `shm_size` (the argv's `--shm-size`), `privileged`
528
+ * (the compat API sends its own, on both), `env_host` (the service's environment did not reach the job), `umask` (the
529
+ * job's umask stayed 0022) and `http_proxy` (a proxy in the service's environment reached no job, with the key absent,
530
+ * true or false): nothing the job saw changed. Documented, not refused.
531
+ */
532
+ export const PODMAN_ROOTFUL_INERT_KEYS = Object.freeze(["pasta_options", "network_cmd_options", "network_cmd_path", "default_capabilities", "no_new_privileges", "init", "oom_score_adj", "pids_limit", "shm_size", "privileged", "env_host", "umask", "http_proxy"]);
533
+
534
+ /**
535
+ * The keys MEASURED REACHING a job, on both venues and both hosts, that move nothing the worker's argv sets as a boundary
536
+ * (issue #448, gate round 1 of PR #473), so they are documented and not refused: `tz` (the job's clock read the named
537
+ * zone) and `no_hosts` (Podman wrote no /etc/hosts into the job, so it has less, not more). Refusing them would refuse
538
+ * a harmless preference, and a refusal that fires on harmless settings stops being read.
539
+ */
540
+ export const PODMAN_HARMLESS_KEYS = Object.freeze(["tz", "no_hosts"]);
541
+
542
+ /**
543
+ * The closed list of observations a backend's `observedBy` may name, each with what it means. Closed for the
544
+ * reason `PROPERTIES` is: a typo in a table entry must not become an observation nobody makes, which would
545
+ * degrade a word forever with nothing saying why.
546
+ */
547
+ export const OBSERVATIONS = Object.freeze({
548
+ [DOCKER_ENDPOINT_LOCAL]: "the docker endpoint this host's docker CLI resolves is on this host",
549
+ [DAEMON_APPLIES_BOUNDS]: "the daemon reports that it applies a container's pid and memory bounds, and is neither rootless nor Podman",
550
+ [RUNTIME_ADDS_NO_MOUNTS]: "the container runtime adds no mounts of its own to a job container",
551
+ [PODMAN_BOUNDS_DELEGATED]: "this worker's rootless Podman runs on cgroup v2 with a systemd user manager running for its account, the pids, memory and cpu controllers delegated to that manager, and Podman putting containers under it, and no containers.conf it reads sets `cgroups`",
552
+ [PODMAN_ADDS_NO_MOUNTS]: "this worker's rootless Podman adds no mounts of its own to a job container",
553
+ [PODMAN_SERVICE_LOCAL]: "the podman CLI this worker spawns runs containers on this host, not through a remote service",
554
+ });
555
+
556
+ /** What to do about each observation that a floor needed and did not get, keyed like `OBSERVATIONS` and pinned to it. */
557
+ export const OBSERVATION_FIX = Object.freeze({
558
+ [DOCKER_ENDPOINT_LOCAL]: "Point the docker CLI back at this host (DOCKER_HOST, DOCKER_CONTEXT or `docker context use`), or lower that entry to `asserted` if the redirect is deliberate.",
559
+ [DAEMON_APPLIES_BOUNDS]: "Run jobs on a rootful Docker Engine that reports PidsLimit and MemoryLimit, or lower that entry to `asserted`: on Podman and rootless daemons `pi-dispatch doctor --live` reads pids.max and memory.max off a real container instead.",
560
+ [RUNTIME_ADDS_NO_MOUNTS]: "On Podman, create an empty /etc/containers/mounts.conf and remove any `volumes` or `mounts` key from containers.conf (and write it in plain ASCII with no escaped key or multi-line string, which the check refuses rather than guesses at), then restart podman.service if it is running, since it keeps the containers.conf it started with, or lower that entry to `asserted`.",
561
+ [PODMAN_BOUNDS_DELEGATED]: "Turn on linger for the worker account (`loginctl enable-linger <account>`, which keeps its systemd user manager running with no one logged in) and run the worker as a systemd user service (`pi-dispatch service install`), or give Podman the account's user bus so it uses the systemd cgroup manager; delegate the cpu, memory and pids controllers to that manager on cgroup v2 (a systemd `Delegate=` drop-in for user@.service), remove any `cgroups` key from its containers.conf (written in plain ASCII with no escaped key or multi-line string, which the check refuses rather than guesses at), or lower that entry to `asserted`: `pi-dispatch doctor --live` reads pids.max and memory.max off a real container.",
562
+ [PODMAN_ADDS_NO_MOUNTS]: "As the worker account, create an empty ~/.config/containers/mounts.conf (or, when it has none, an empty /etc/containers/mounts.conf), remove any `volumes`, `mounts`, `devices` or `hooks_dir` key from every containers.conf it reads (and write those in plain ASCII with no escaped key or multi-line string, which the check refuses rather than guesses at) and any OCI hook, or lower that entry to `asserted`.",
563
+ [PODMAN_SERVICE_LOCAL]: "Unset CONTAINER_HOST and any containers.conf service destination for the worker account so `podman info` reports serviceIsRemote false, or lower that entry to `asserted`.",
564
+ });
565
+
274
566
  const BACKENDS_TABLE = {
275
567
  local: {
276
568
  /**
@@ -285,15 +577,20 @@ const BACKENDS_TABLE = {
285
577
  // against the imported array by two tests. `dockerArgsFromSpec` additionally refuses a
286
578
  // `dockerExtra` carrying a flag that would supersede one of them -- membership in an argv is not
287
579
  // effectiveness of that argv, and without that guard those two assertions would pass on an argv
288
- // with no boundary left. It is a deny-list, so it NARROWS that gap rather than closing it; what
289
- // closes it today is that the only production caller passes fixed literals.
580
+ // with no boundary left. The deny-list only NARROWED that gap; since issue #341 an allow-list of what
581
+ // the callers pass (`DOCKER_EXTRA_ALLOWED`) closes it, and every caller still passes fixed literals.
582
+ //
583
+ // AND ONLY WHILE OBSERVED (issue #345): the pid and memory bounds in those flags are the daemon's to apply, and a
584
+ // rootless daemon without cgroup delegation accepts `--pids-limit` and applies nothing (measured). So the word is
585
+ // earned by `daemonAppliesBounds` below, and degrades to `asserted` where the daemon is not observed applying them.
290
586
  isolation: ENFORCED,
291
587
  // `--rm` leads ISOLATION_FLAGS and the container name carries the job id, so no container is
292
588
  // reachable to reuse even in principle.
293
589
  ephemeral: ENFORCED,
294
590
  // The mount list is built by `containerSpec` from a fixed set of named host paths and nothing
295
591
  // else. There is no pass-through, and the shared session store under PI_SESSIONS_DIR is never
296
- // among them -- only this job's own copy.
592
+ // among them -- only this job's own copy. AND ONLY WHILE OBSERVED (issue #345): a runtime can mount things of
593
+ // its own that no argv names (rootful Podman's `/run/secrets`), so the word is earned by `runtimeAddsNoMounts`.
297
594
  mountSet: ENFORCED,
298
595
  // CAPABILITY, gated by PI_EGRESS. When armed: `--network=pi-job-<id>-net`, `--internal`, created
299
596
  // before the spawn and removed in a finally. When PI_EGRESS=0 the flag is absent and the job is
@@ -324,15 +621,23 @@ const BACKENDS_TABLE = {
324
621
  // and `buildRecord` is an explicit literal with no spread. Where the values GO is
325
622
  // `credentialTransit`'s question, not this one's.
326
623
  secretsCustody: ENFORCED,
327
- // ASSERTED, and this is the second honest one. The intent is that the daemon is on this host, so
328
- // nothing leaves it -- but every spawn is `docker` with the worker's own environment inherited,
329
- // and `DOCKER_HOST=tcp://...` or a `docker context` redirects that connection to another machine
330
- // with the provider key and the per-job forge token riding along as `-e NAME=VALUE`. `DOCKER_HOST`
331
- // appears NOWHERE in this repository: no code sets it, no check refuses it, no test reads it back.
332
- // By this file's own definition of `enforced` -- "this worker CAN build it, in its own code, and a
333
- // test in this repo reads it back" -- there is no such code, so the word would be exactly the
334
- // overclaim `nonRoot` avoids one property up. A boot check on DOCKER_HOST would earn `enforced`.
335
- credentialTransit: ASSERTED,
624
+ // ENFORCED, AND ONLY WHILE OBSERVED (issue #278). Every spawn is `docker` with the worker's own
625
+ // environment inherited, and `DOCKER_HOST`, `DOCKER_CONTEXT` or a context selected in the CLI's config
626
+ // can send that connection to another machine with the provider key and the per-job forge token riding
627
+ // along as `-e NAME=VALUE`. This was `asserted` while nothing in the repository looked. Now the worker
628
+ // asks the CLI which endpoint it resolves (`backend-local.mjs`, `makeDockerEndpointResolver`) at boot
629
+ // and before every job's spend, and the word is what that answer earns: `observedBy` below names the
630
+ // observation, and `effectiveWord` degrades this to `asserted` -- asserted by the operator who pointed
631
+ // the CLI elsewhere -- whenever the endpoint is not observed on this host. A floor asking for
632
+ // `enforced` then refuses; a floor asking for `asserted` holds, because pointing the worker at a daemon
633
+ // is the operator's call. That split is `unarmedFloor`'s lesson: a floor is not met by capability alone.
634
+ //
635
+ // WHAT THE OBSERVATION CANNOT SEE, stated rather than claimed away: it judges the endpoint's FORM, so a
636
+ // unix socket or a loopback port that is really a tunnel (`ssh -L`, socat) to another machine reads as
637
+ // local, as does a `localhost` this host's resolver maps elsewhere; and a redirect that lands between the per-job read and that job's `docker run` spawn is not
638
+ // caught for that job. That window is not small: the read sits before the spend, so it spans the secret
639
+ // resolution, the token mint, the clone and the reservation.
640
+ credentialTransit: ENFORCED,
336
641
  // The whole reason `DES-WORKER-ON-HOST` reversed the containerised worker.
337
642
  localFolders: ENFORCED,
338
643
  },
@@ -346,12 +651,101 @@ const BACKENDS_TABLE = {
346
651
  * rather than aspirational. For a vendor adapter this is where "the vendor's documentation" goes.
347
652
  */
348
653
  asserts: {
349
- nonRoot: "the job image's USER directive (this repo's builds `USER pi`; an operator-built image may not)",
350
- credentialTransit: "the docker endpoint DOCKER_HOST resolves to, which is this host unless something redirects it",
654
+ // Issue #341: on a daemon that enforces bind-mount ownership the argv supplies the uid instead, validated
655
+ // non-zero by the builder; the word stays asserted because on Docker Desktop, and for a uid-1001 worker, the
656
+ // image still provides it.
657
+ nonRoot: "the job image's USER directive (this repo's builds `USER pi`; an operator-built image may not), or, on a daemon that enforces bind-mount ownership with a worker uid other than 1001, the worker's own non-zero uid passed as `--user`",
658
+ },
659
+ /**
660
+ * Which declared words hold only while something about THIS HOST is observed (issue #278), as
661
+ * `{ property: observation }`. A per-BACKEND map rather than a field on `PROPERTIES` like `armedBy`:
662
+ * `armedBy` names a deployment switch that means the same thing for every venue, while an observation of
663
+ * this host's docker CLI means nothing for a remote venue, whose `credentialTransit` a vendor might assert.
664
+ */
665
+ observedBy: {
666
+ isolation: DAEMON_APPLIES_BOUNDS,
667
+ mountSet: RUNTIME_ADDS_NO_MOUNTS,
668
+ credentialTransit: DOCKER_ENDPOINT_LOCAL,
669
+ },
670
+ },
671
+ podman: {
672
+ /**
673
+ * The worker account's own ROOTLESS Podman, driven through the real `podman` CLI (issue #354,
674
+ * `DES-PODMAN-NATIVE-ROOTLESS-BACKEND`). Not `local` pointed at Podman: that route goes through Podman's Docker
675
+ * API and keeps `local`'s words. This one runs `podman run` itself, always as the worker's own `<euid>:<egid>`
676
+ * under `--userns=keep-id`, and REFUSES what it was not measured on: a rootful Podman (keep-id there is not
677
+ * refused, and adds supplementary group 0, measured), a remote service, a root worker, and any platform but
678
+ * Linux (`backend-podman.mjs`, `decidePodmanJobUser`). Every word below was measured on rootless Podman 5.8.1,
679
+ * Fedora 44, 2026-09-25.
680
+ */
681
+ describe: "this worker account's own rootless Podman, through the podman CLI",
682
+ // The containers run on this host, as this account. A remote service is refused rather than declared remote, so
683
+ // the venue is never one; `credentialTransit`'s observation is what notices a CLI that tries.
684
+ remote: false,
685
+ declares: {
686
+ // The same `ISOLATION_FLAGS` through the same builder (`buildPodmanRunArgs` shares `argsFromSpec`), read back on
687
+ // a live rootless container: CapEff 0, NoNewPrivs 1, seccomp 2, pids.max 512, memory.max 4 GiB. AND ONLY WHILE
688
+ // OBSERVED: the bounds are cgroup v2's, and a user without the controllers delegated accepts the flags and
689
+ // applies nothing, so the word is earned by `podmanBoundsDelegated`.
690
+ isolation: ENFORCED,
691
+ // `--rm` leads ISOLATION_FLAGS and the name carries the job id; measured, `--rm` removes the container (and its
692
+ // cidfile) when it exits.
693
+ ephemeral: ENFORCED,
694
+ // The same `containerSpec` mount list. AND ONLY WHILE OBSERVED: rootless Podman mounts what the winning
695
+ // mounts.conf lists (`/run/secrets` on Fedora and RHEL by default), invisible in `.Mounts`, so the word is earned
696
+ // by `podmanAddsNoMounts`. Nothing is relabelled but the job's own per-job directories (`:Z`).
697
+ mountSet: ENFORCED,
698
+ // CAPABILITY, gated by PI_EGRESS, as on `local`: a per-job `--internal` network and the proxy attached to it.
699
+ // Measured rootless: an `--internal` network reaches NOTHING on the host (its IP, host.containers.internal,
700
+ // 10.0.2.2 and the gateway all fail), and a proxy started on a NAMED bridge network under the same rootless
701
+ // Podman can be `network connect`ed to it (a pasta or slirp4netns one cannot: exit 125).
702
+ egress: ENFORCED,
703
+ // Same switch: the job and the proxy are the only endpoints on an `--internal` network. Measured, a peer on
704
+ // another job's network is unreachable (ENOTFOUND / ENETUNREACH).
705
+ jobToJobIsolation: ENFORCED,
706
+ // `--pull=never` in ISOLATION_FLAGS; measured, an absent image exits 125 "image not known" with no lookup.
707
+ imagePinning: ENFORCED,
708
+ // `spawn` on the CLI, and a payload's exit N reaches `close` as N (measured). The one residual, named in the
709
+ // design entry and not normalised: with `--init`, an entrypoint catatonit cannot exec exits 1, which only its
710
+ // stderr tells from a payload's own exit 1, so it retries as infra.
711
+ exitCodes: ENFORCED,
712
+ // `podman stop -t 5` on the name, and the abort FLAG is the discriminator, as on `local`.
713
+ abortable: ENFORCED,
714
+ // `-v <host>:/job:ro`: the kernel's, and the bundle says it binds.
715
+ readOnlyJobInputs: ENFORCED,
716
+ // ENFORCED here where `local` can only assert it, because the ARGV supplies the uid on every job: `--user
717
+ // <euid>:<egid>` beside `--userns=keep-id`, which the builder refuses to emit without a user and refuses for a
718
+ // uid or gid of 0 (`assertJobUser`). A root worker and a rootful Podman are refused before any spawn, so the uid
719
+ // inside is the worker's own unprivileged one on the host (measured rootless: runs as 1234, no root group).
720
+ nonRoot: ENFORCED,
721
+ // The same resolver, record and comment code as `local`; where the values go is `credentialTransit`'s.
722
+ secretsCustody: ENFORCED,
723
+ // ENFORCED, AND ONLY WHILE OBSERVED, for `local`'s reason with Podman's knob: `CONTAINER_HOST`, `--remote` or a
724
+ // service destination in containers.conf send the run, and its `-e` values, to a service elsewhere. The worker
725
+ // asks `podman info` whether the service is remote, and the word is earned by `podmanServiceLocal`.
726
+ credentialTransit: ENFORCED,
727
+ // A bind mount of the operator's folder, as on `local`, never relabelled (on an SELinux host it needs the
728
+ // `semanage fcontext` doctor names), and writable because the job runs as the account that owns it.
729
+ localFolders: ENFORCED,
730
+ },
731
+ // Nothing asserted: every word above is this worker's own argv or an observation of this host.
732
+ asserts: {},
733
+ // Podman's OWN observations, never `local`'s: those read `docker info` and the docker CLI's endpoint, which say
734
+ // nothing about the podman CLI's rootless store.
735
+ observedBy: {
736
+ isolation: PODMAN_BOUNDS_DELEGATED,
737
+ mountSet: PODMAN_ADDS_NO_MOUNTS,
738
+ credentialTransit: PODMAN_SERVICE_LOCAL,
351
739
  },
352
740
  },
353
741
  };
354
742
 
743
+ /**
744
+ * The podman venue's name (issue #354): the table key, the bundle's `name` and what `PI_BACKENDS` spells, one constant so
745
+ * the adapter, the boot wiring and doctor cannot disagree about it.
746
+ */
747
+ export const PODMAN_BACKEND = "podman";
748
+
355
749
  /**
356
750
  * FROZEN, deeply. `makeLocalBackend` hands `BACKENDS[name].declares` out on the bundle it returns, so
357
751
  * without this a consumer holding a backend could assign one word and silently rewrite what every later
@@ -360,15 +754,47 @@ const BACKENDS_TABLE = {
360
754
  */
361
755
  export const BACKENDS = Object.freeze(
362
756
  Object.fromEntries(
363
- Object.entries(BACKENDS_TABLE).map(([name, entry]) => [name, Object.freeze({ ...entry, declares: Object.freeze({ ...entry.declares }) })]),
757
+ // EVERY nested map, not just `declares`: `asserts` is what `doctor` prints beside a word and `observedBy` is
758
+ // what decides whether the word holds, and an unfrozen one is the shared mutable declaration this freeze
759
+ // exists to prevent (#278 found `asserts` had been left mutable).
760
+ Object.entries(BACKENDS_TABLE).map(([name, entry]) => [
761
+ name,
762
+ Object.freeze({ ...entry, declares: Object.freeze({ ...entry.declares }), asserts: Object.freeze({ ...(entry.asserts ?? {}) }), observedBy: Object.freeze({ ...(entry.observedBy ?? {}) }) }),
763
+ ]),
364
764
  ),
365
765
  );
366
766
 
367
767
  export const BACKEND_NAMES = Object.freeze(Object.keys(BACKENDS));
368
768
 
369
- /** The default, and the name a deployment that has never heard of this table is running. */
769
+ /**
770
+ * The default, and the name a deployment that has never heard of this table is running: what `PI_BACKENDS`
771
+ * means when it is unset. NOT "a venue every deployment holds": since issue #354 a deployment may bless a list
772
+ * without `local`, and its default is then that list's first entry (`config.defaultBackend`). Code that means
773
+ * the local adapter by name, rather than the unset default, compares against this same word on purpose and
774
+ * says so where it does.
775
+ */
370
776
  export const DEFAULT_BACKEND = "local";
371
777
 
778
+ /**
779
+ * The venue an artifact that NAMES NO VENUE was produced in (issue #277): a session transcript with no stamp
780
+ * beside it, or a sandbox manifest with no `backend` key. Both are written with a venue from #277 on, so one
781
+ * without was written before venues were recorded, when `local` was the only entry this table had -- a fact
782
+ * about the past rather than a setting.
783
+ *
784
+ * DELIBERATELY NOT THE DEPLOYMENT DEFAULT, though today the two spell the same word. The default is
785
+ * `PI_BACKENDS[0]`, which a deployment can change; reading an unstamped transcript as "whatever the default
786
+ * is now" would hand a local transcript to a remote venue the day `PI_BACKENDS` lists that venue first. The
787
+ * past does not move when the configuration does.
788
+ *
789
+ * STILL `local` ON A HOST THAT NO LONGER BLESSES IT (issue #354), and that is the rule rather than an oversight.
790
+ * Such a host's default is another venue, so an unstamped transcript there matches no venue it runs and the key
791
+ * cold-starts once as `venue-changed`, then carries the new venue's stamp. Resolving absence to the default
792
+ * instead would resume a Docker-written transcript in a container another runtime built, on the strength of a
793
+ * stamp that was never written: the one thing the stamp exists to refuse. A sandbox manifest with no key is
794
+ * likewise a `local` run, reopened only through the local adapter's own CLI.
795
+ */
796
+ export const UNATTRIBUTED_BACKEND = "local";
797
+
372
798
  /**
373
799
  * One backend's entry, or `undefined`. The caller decides how loudly an unknown name fails.
374
800
  *
@@ -416,7 +842,8 @@ export function shortfall(name, want = {}) {
416
842
 
417
843
  /**
418
844
  * `PI_BACKENDS` -- which backends this deployment blesses, comma separated. Unset means `[local]`, which is
419
- * what every deployment that has never heard of this table is already running.
845
+ * what every deployment that has never heard of this table is already running. Set, any non-empty list of
846
+ * known names is a deployment, `local` or not; its first entry is the default.
420
847
  *
421
848
  * ENV-ONLY, never the settings overlay and never the deployment pointer, on `config.mjs`'s rule for
422
849
  * `PI_SECRET_RESOLVER_ROOTS`: "a bound that can be widened from the surface it bounds is not a bound". The
@@ -430,26 +857,51 @@ export function shortfall(name, want = {}) {
430
857
  * Throws a plain Error; `config.mjs` re-tags it as a config error, which is `egressArmed`'s arrangement and
431
858
  * for its reason: this module imports nothing, so it cannot reach for that tagger itself.
432
859
  */
433
- export function parseBackendList(raw) {
860
+ export function parseBackendList(raw, { known = BACKEND_NAMES } = {}) {
434
861
  const names = (raw ?? "")
435
862
  .split(",")
436
863
  .map((s) => s.trim())
437
864
  .filter((s) => s.length > 0);
438
865
  if (names.length === 0) return [DEFAULT_BACKEND];
439
866
  for (const name of names) {
440
- if (!Object.hasOwn(BACKENDS, name)) {
441
- throw new Error(`PI_BACKENDS names an unknown backend ${JSON.stringify(name)} (known: ${BACKEND_NAMES.join(", ")})`);
867
+ // `known` is a TEST seam: it was added while `local` was the table's one entry, when a list without it could not
868
+ // be written through the real table (since issue #354 `podman` can). Every production caller passes nothing and
869
+ // gets the table's own names.
870
+ if (!known.includes(name)) {
871
+ throw new Error(`PI_BACKENDS names an unknown backend ${JSON.stringify(name)} (known: ${known.join(", ")})`);
442
872
  }
443
873
  }
444
- // Deduplicated, order preserved: the first entry is what a deployment means by "the default one".
445
- const unique = [...new Set(names)];
446
- // The DEFAULT backend must stay in the set. A trigger that names no venue is dispatched to
447
- // `backends[0]`, and the boot registry refuses a default it does not hold -- so a set excluding it would
448
- // describe a deployment whose unflagged triggers, which is nearly all of them, have nowhere to run.
449
- if (!unique.includes(DEFAULT_BACKEND)) {
450
- throw new Error(`PI_BACKENDS must include ${JSON.stringify(DEFAULT_BACKEND)}: a trigger that names no backend is dispatched there, so a set without it leaves every unflagged trigger nowhere to run`);
874
+ // Deduplicated, order preserved: the first entry is what a deployment means by "the default one", and it is
875
+ // where a trigger that names no venue is dispatched.
876
+ //
877
+ // `local` IS NOT REQUIRED (issue #354, the operator's decision: `PI_BACKENDS=podman` alone is a deployment). The
878
+ // guard that stood here refused any set without it, on the reasoning that an unflagged trigger is dispatched to
879
+ // the default and the default had to be `local`. It never had to: unflagged triggers go to `backends[0]`,
880
+ // whatever it names, and the boot registry refuses a default it does not hold, so a set whose first entry is
881
+ // built is runnable with or without `local` in it. What the guard actually did was force a host with no Docker
882
+ // to spawn `docker` at every boot for a venue it would never use.
883
+ return [...new Set(names)];
884
+ }
885
+
886
+ /**
887
+ * Which venues run jobs here (issue #354), from the same parse the worker boots with. An unparseable PI_BACKENDS reads as
888
+ * the unset default, `local` alone, so every docker line is exactly what it always was and only the backend section,
889
+ * which reports the parse, fails on it: guessing another venue from a list the worker refuses would be doctor inventing
890
+ * a deployment.
891
+ *
892
+ * Lives here rather than in doctor (where it started) since issue #430: `up` asks the same question to decide whether to
893
+ * drive docker, podman or both, and two copies of "which venues does this list bless" is two answers to one question.
894
+ * A caller that must REFUSE on an unparseable list (`service install`, which would otherwise install a docker-shaped
895
+ * unit for a deployment that meant podman) calls `parseBackendList` itself first.
896
+ */
897
+ export function venuesOf(env) {
898
+ let blessed;
899
+ try {
900
+ blessed = parseBackendList(env?.PI_BACKENDS);
901
+ } catch {
902
+ blessed = [DEFAULT_BACKEND];
451
903
  }
452
- return unique;
904
+ return { localUsed: blessed.includes(DEFAULT_BACKEND), podmanUsed: blessed.includes(PODMAN_BACKEND), podmanDefault: blessed[0] === PODMAN_BACKEND };
453
905
  }
454
906
 
455
907
  /**