@edgehero/pi-dispatch 1.10.3 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.example +303 -150
  2. package/README.md +52 -0
  3. package/deploy/com.pi-dispatch.worker.plist +10 -4
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +12 -1
  15. package/deploy/worker-env-wrapper.sh +63 -37
  16. package/deploy/worker.service +18 -8
  17. package/package.json +15 -5
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4756 -394
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +456 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +245 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-pi.mjs +19 -3
  54. package/src/host-registry.mjs +29 -2
  55. package/src/identity.mjs +29 -4
  56. package/src/image-preflight.mjs +46 -11
  57. package/src/image-ref.mjs +21 -0
  58. package/src/index.mjs +363 -13
  59. package/src/init.mjs +197 -38
  60. package/src/job-user.mjs +252 -0
  61. package/src/json-duplicates.mjs +204 -0
  62. package/src/live-probes.mjs +1020 -0
  63. package/src/materialize.mjs +4 -11
  64. package/src/netns-keeper.mjs +264 -0
  65. package/src/on-failure.mjs +119 -0
  66. package/src/outbox.mjs +7 -0
  67. package/src/packages.mjs +2 -2
  68. package/src/podman-stack.mjs +1304 -0
  69. package/src/prepare-github.mjs +6 -6
  70. package/src/prepare-local.mjs +51 -17
  71. package/src/prepare.mjs +27 -6
  72. package/src/pricing.mjs +9 -5
  73. package/src/processor.mjs +506 -26
  74. package/src/provider-key.mjs +66 -0
  75. package/src/provider-steering.mjs +185 -0
  76. package/src/queue.mjs +35 -8
  77. package/src/redact.mjs +84 -0
  78. package/src/reserved-env.mjs +7 -3
  79. package/src/retention-sweep.mjs +178 -0
  80. package/src/run-container.mjs +181 -14
  81. package/src/run-history.mjs +105 -16
  82. package/src/runtime-observations.mjs +1152 -0
  83. package/src/runtime-settings.mjs +13 -8
  84. package/src/sandbox-cli.mjs +100 -95
  85. package/src/sandbox-store.mjs +612 -45
  86. package/src/sandbox.mjs +1459 -37
  87. package/src/schedules.mjs +16 -3
  88. package/src/secret-profiles.mjs +2 -1
  89. package/src/secrets.mjs +24 -6
  90. package/src/service-env.mjs +247 -0
  91. package/src/service.mjs +618 -28
  92. package/src/session-store.mjs +678 -53
  93. package/src/start.mjs +1348 -326
  94. package/src/subscriptions.mjs +7 -3
  95. package/src/transient.mjs +240 -0
  96. package/src/triggers-file.mjs +71 -15
  97. package/src/triggers.mjs +179 -19
  98. package/src/up.mjs +1399 -85
  99. package/src/valkey-auth.mjs +529 -0
  100. package/src/valkey-endpoint.mjs +367 -0
  101. package/src/watch-closer.mjs +158 -0
package/src/service.mjs CHANGED
@@ -41,11 +41,19 @@
41
41
  * the rest of the config is broken.
42
42
  */
43
43
  import { spawn as nodeSpawn } from "node:child_process";
44
- import { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync } from "node:fs";
45
- import { homedir, tmpdir, userInfo } from "node:os";
44
+ import { chmodSync, existsSync, mkdirSync, readFileSync, realpathSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs";
45
+ import { lookup as dnsLookup } from "node:dns/promises";
46
+ import { connect as netConnect } from "node:net";
47
+ import { homedir, networkInterfaces, tmpdir, userInfo } from "node:os";
46
48
  import { dirname, join, resolve } from "node:path";
47
49
  import { fileURLToPath } from "node:url";
48
50
  import { parseArgs } from "node:util";
51
+ import { parseBackendList, venuesOf } from "./backends.mjs";
52
+ import { sharedShellIgnored } from "./deployment-venue.mjs";
53
+ import { egressArmed, egressProxyName } from "./egress.mjs";
54
+ import { updateEnvFile } from "./env-file.mjs";
55
+ import { VALKEY_HEALTH_SCRIPT, VALKEY_PASSWORD_KEY, VALKEY_START_SCRIPT, dollarsDoubled, newValkeyPassword, valkeyPasswordDecision } from "./valkey-auth.mjs";
56
+ import { ALL_QUADLET_FILES, ALLOWLIST_PLACEHOLDER, NETNS_KEEPER, NETNS_KEEPER_FORMAT, judgeNetnsKeeper, keeperUnderRunningProxyHint, managerEnvRefusal, PROXY_CONF_PLACEHOLDER, QUADLET_FILES, applyStack, decideValkey, describeAction, passwdNameFrom, readSubuidRanges, readValkeyKeys, valkeySharedOn, VALKEY_SHARED_KEY, describeRollBack, journalWrite, rollBackWrites, foreignContainerRefusal, foreignContainers, lingerNote, planStack, proxyConfCopyPath, proxyRestartWarning, quadletDir, readLinger, readStackKeys, stackComponents, unknownContainerRefusal, userBusRefusal, valkeyEnvPath, valkeyPasswordRestartWarning, workerUnitDeps } from "./podman-stack.mjs";
49
57
 
50
58
  // src/ is where this module lives in BOTH layouts (worker/src in a checkout,
51
59
  // node_modules/@edgehero/pi-dispatch/src under npm). Deploy templates resolve one level up from it
@@ -74,6 +82,10 @@ function resolveReceiverStart() {
74
82
  * worker/test/service.test.mjs asserts each one is still present in the real deploy/ file, so an edit
75
83
  * to a template that would break the render fails the build instead of shipping a broken `service`.
76
84
  */
85
+ // The two worker-template literals a podman render rewrites (issue #430), spelled once for the pin table and the render.
86
+ const WORKER_DESCRIPTION = "Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)";
87
+ const WORKER_VALKEY_COMMENT = "# Valkey must be reachable (on this host, `pi-dispatch up` in the deployment folder starts it), but it is a separate\n# unit/container -- not ordered here since it may be remote.\n";
88
+
77
89
  export const TEMPLATE_PINS = {
78
90
  "worker.service": [
79
91
  "ExecStart=/usr/bin/node worker/src/cli.mjs worker", // the WHOLE line → `<execPath> <cliPath> worker` (cli.mjs sits beside this module in src/ in both layouts)
@@ -81,6 +93,9 @@ export const TEMPLATE_PINS = {
81
93
  "EnvironmentFile=/opt/pi-dispatch/.env", // → <deployDir>/.env — the operator's .env lives beside the units, never inside the package
82
94
  "\nUser=pi\n", // the DIRECTIVE line (the header comment also says User=pi mid-line, hence the \n anchors): stripped for --user scope; rewritten to the invoking user for --system
83
95
  "WantedBy=multi-user.target", // → default.target in user scope (multi-user.target never runs there)
96
+ "\nWants=network-online.target\n", // the anchor the podman venue's Wants=/After= on its Quadlet units go after (issue #430), user scope only
97
+ "Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)", // → names rootless podman on a podman deployment (issue #430)
98
+ "# Valkey must be reachable (on this host, `pi-dispatch up` in the deployment folder starts it), but it is a separate\n# unit/container -- not ordered here since it may be remote.\n", // → says it IS ordered, when the Quadlet Valkey is installed
84
99
  // Byte-for-byte survivors — semantics the render must not lose:
85
100
  "RestartPreventExitStatus=2", // EXIT_POLICY is never restarted (a retry loop is a bill)
86
101
  "StartLimitIntervalSec=60",
@@ -132,6 +147,52 @@ export const TEMPLATE_PINS = {
132
147
  // a unit that starts fine and ignores the operator's secrets manager.
133
148
  "worker-env-wrapper.sh": ["PI_ENV_SETUP"],
134
149
  "worker-env-wrapper.cmd": ["PI_ENV_SETUP"],
150
+ // The podman venue's stack as Quadlet units (issue #430, podman-stack.mjs). Three are copied verbatim, so what is
151
+ // pinned is what the worker and the installer rely on by NAME: the container and network names the worker attaches
152
+ // by, the unit names the worker unit Wants=, the loopback-only port, and an explicit Network= in every .container
153
+ // (a containers.conf `netns = "host"` would otherwise put it in the host namespace, measured).
154
+ "pi-dispatch-valkey.network": ["NetworkName=pi-dispatch-valkey"],
155
+ "pi-dispatch-valkey.container": [
156
+ "ContainerName=pi-dispatch-valkey",
157
+ "Network=pi-dispatch-valkey.network",
158
+ "PublishPort=127.0.0.1:6379:6379",
159
+ "Volume=pi-dispatch-valkey-data:/data",
160
+ // Issue #468: the password from a 0600 file into the container's environment, then to valkey-server as
161
+ // configuration on stdin, never on an argv (tini keeps its argv as PID 1, readable by every account). The two lines
162
+ // are pinned to the ONE copy of each script in valkey-auth.mjs, `$` written `$$` for systemd.
163
+ "EnvironmentFile=%h/.config/pi-dispatch/valkey.env",
164
+ `Exec=sh -c '${dollarsDoubled(VALKEY_START_SCRIPT)}'`,
165
+ `HealthCmd=${dollarsDoubled(VALKEY_HEALTH_SCRIPT)}`,
166
+ "WantedBy=default.target",
167
+ ],
168
+ "pi-dispatch-egress-out.network": ["NetworkName=pi-dispatch-egress-out"],
169
+ "pi-dispatch-egress-proxy.container": [
170
+ `Volume=${PROXY_CONF_PLACEHOLDER}:/etc/squid/squid.conf:ro,z`, // → the account-owned COPY of the package's egress-proxy.conf (~/.config/pi-dispatch/egress-proxy.conf, podman-stack.mjs proxyConfCopyPath), because `z` cannot relabel a root-owned package file (measured)
171
+ `Volume=${ALLOWLIST_PLACEHOLDER}:/etc/pi-dispatch/allowlist.conf:ro,z`, // → <deployDir>/egress-allowlist.conf, the list `init` scaffolds
172
+ "ContainerName=pi-dispatch-egress-proxy",
173
+ "Network=pi-dispatch-egress-out.network",
174
+ "WantedBy=default.target",
175
+ ],
176
+ // The rootless network keeper (issue #458), copied verbatim. Pinned: the names (outside every sweep's prefix, and
177
+ // what doctor reads), its own network with no route and no DNS, and every line that keeps it from widening anything
178
+ // (podman-stack.test.mjs reads the same lines back as the argv they generate).
179
+ "pi-dispatch-netns-keeper.network": ["NetworkName=pi-dispatch-netns-keeper", "Internal=true", "DisableDNS=true"],
180
+ "pi-dispatch-netns-keeper.container": [
181
+ "ContainerName=pi-dispatch-netns-keeper",
182
+ "Network=pi-dispatch-netns-keeper.network",
183
+ "ReadOnly=true",
184
+ "DropCapability=all",
185
+ "NoNewPrivileges=true",
186
+ "User=65534",
187
+ "Group=65534",
188
+ "RunInit=true",
189
+ "ExecStartPre=/usr/bin/podman network create --ignore --disable-dns --internal pi-dispatch-netns-keeper",
190
+ "Restart=always",
191
+ "RestartSec=1s",
192
+ "StartLimitIntervalSec=0",
193
+ "SuccessExitStatus=143",
194
+ "WantedBy=default.target",
195
+ ],
135
196
  };
136
197
 
137
198
  /**
@@ -231,6 +292,19 @@ export function readUnitSeam(text, platform) {
231
292
  return { setup: clip(grab(readers.setup)), deployDir: clip(grab(readers.deployDir)) };
232
293
  }
233
294
 
295
+ /**
296
+ * The account a SYSTEM unit runs the worker as (`User=`), or `null` when it names none (issue #341). Only systemd has
297
+ * one: a user-scope unit, a launchd agent and an nssm service run as whoever installed them. Separate from
298
+ * `readUnitSeam` because it is not part of the render round trip (a `--system` render writes the invoking user, a
299
+ * `--user` render strips the line). Read as systemd does: whitespace around `=` allowed, and the LAST assignment wins.
300
+ */
301
+ export function readUnitUser(text, platform) {
302
+ if (platform !== "linux" || typeof text !== "string") return null;
303
+ const hits = [...text.replace(/\0/g, "").matchAll(/^[ \t]*User[ \t]*=[ \t]*(.*?)[ \t]*\r?$/gm)];
304
+ const value = hits.at(-1)?.[1];
305
+ return value ? value : null;
306
+ }
307
+
234
308
  const SUBCOMMANDS = new Set(["render", "install", "uninstall", "status", "start", "stop", "restart"]);
235
309
 
236
310
  const SERVICE_USAGE = `pi-dispatch service — run the worker (or --receiver) as an OS service, rendered for THIS host
@@ -269,13 +343,26 @@ export async function runService(argv = [], deps = {}) {
269
343
  // this command. An operator changes it by running the command as someone else, not by declaring it.
270
344
  user = env.USER || userInfo().username,
271
345
  tmp = tmpdir(),
272
- fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync },
346
+ fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync, chmodSync, statSync, renameSync, realpathSync },
273
347
  spawn = nodeSpawn,
274
348
  out = (s) => process.stdout.write(s),
275
349
  err = (s) => process.stderr.write(s),
276
350
  sleep = (ms) => new Promise((r) => setTimeout(r, ms)),
277
351
  now = () => Date.now(),
278
352
  queue = null, // test seam; production builds one lazily in doRestart from VALKEY_URL
353
+ // Is anything on 127.0.0.1:6379? Asked only on a podman deployment, to decide whether the Quadlet Valkey is
354
+ // wanted (issue #430); the same plain TCP connect `up` uses, for up's reason. Issue #464: asked of every address
355
+ // VALKEY_URL's host resolves to (`lookup`, as the worker's client resolves it), and `interfaces` says which of
356
+ // them are this host's.
357
+ probeTcp = defaultProbeTcp,
358
+ lookup = (host, opts) => dnsLookup(host, opts),
359
+ interfaces = networkInterfaces,
360
+ // Issue #464 (gate round 2): how `getsubids` is run for this account's subordinate ranges (null: not installed).
361
+ runSync = undefined,
362
+ // PR #463: how a path resolves through symlinks, for the manager-environment rule (a symlinked home).
363
+ realpath = (p) => realpathSync(p),
364
+ // Issue #468: how a new Valkey password is made (a test's seam; 32 random bytes, hex, never shown).
365
+ newPassword = newValkeyPassword,
279
366
  } = deps;
280
367
 
281
368
  let values, positionals;
@@ -338,6 +425,12 @@ export async function runService(argv = [], deps = {}) {
338
425
  sleep,
339
426
  now,
340
427
  queue,
428
+ probeTcp,
429
+ lookup,
430
+ interfaces,
431
+ runSync,
432
+ realpath,
433
+ newPassword,
341
434
  which: values.receiver ? "receiver" : "worker",
342
435
  scope: platform === "linux" && values.system ? "system" : "user",
343
436
  force: values.force,
@@ -363,7 +456,7 @@ export async function runService(argv = [], deps = {}) {
363
456
  return doRender(ctx);
364
457
  case "install":
365
458
  // --print is implied for render and opt-in here: see what will be written, then write it.
366
- if (values.print) doRender(ctx);
459
+ if (values.print) await doRender(ctx);
367
460
  return doInstall(ctx);
368
461
  case "uninstall":
369
462
  return doUninstall(ctx);
@@ -511,7 +604,7 @@ function composeEnvSetupExec(setup, argv) {
511
604
  * known literals (see TEMPLATE_PINS); everything else — RestartPreventExitStatus=2, the StartLimit
512
605
  * crash-loop bound, KillSignal, TimeoutStopSec — passes through byte-for-byte.
513
606
  */
514
- function renderLinuxUnit(ctx) {
607
+ function renderLinuxUnit(ctx, stackUnits = [], venues = null) {
515
608
  const template = ctx.which === "receiver" ? "receiver.service" : "worker.service";
516
609
  // The ExecStart line is replaced WHOLE, not path-by-path: the template's script path is relative
517
610
  // to a repo-root WorkingDirectory that only a checkout has. The rendered unit points at absolute
@@ -549,6 +642,19 @@ function renderLinuxUnit(ctx) {
549
642
  // symlink into a .wants/ directory no user-instance boot ever walks — enabled but never
550
643
  // started. default.target is the user manager's boot target.
551
644
  unit = unit.replace("WantedBy=multi-user.target", "WantedBy=default.target");
645
+ // The podman venue's Quadlet units (issue #430), which live in this same user manager, so the worker can be
646
+ // ordered after them. Nothing is added when there are none, which keeps every other render byte-identical.
647
+ const deps = workerUnitDeps(stackUnits);
648
+ if (deps) unit = unit.replace("\nWants=network-online.target\n", () => `\nWants=network-online.target\n${deps}`);
649
+ // The template's own words, made true for a podman deployment (round 2 nit): its Description says jobs launch
650
+ // via docker, and its comment says Valkey is not ordered here. Both anchors are TEMPLATE_PINS.
651
+ if (venues?.podmanUsed) {
652
+ const runtime = venues.localUsed ? "docker and rootless podman" : "rootless podman";
653
+ unit = unit.replace(WORKER_DESCRIPTION, () => `Description=pi-dispatch worker (drains the job queue on the host; launches job containers via ${runtime})`);
654
+ if (stackUnits.includes(QUADLET_FILES.valkey.unit)) {
655
+ unit = unit.replace(WORKER_VALKEY_COMMENT, () => "# Valkey is the podman venue's Quadlet unit in this same user manager, so the worker is ordered after it\n# (the Wants=/After= lines `pi-dispatch service install` added below).\n");
656
+ }
657
+ }
552
658
  } else {
553
659
  unit = unit.replace(/^User=pi$/m, () => `User=${ctx.user}`);
554
660
  }
@@ -558,8 +664,11 @@ function renderLinuxUnit(ctx) {
558
664
  /**
559
665
  * Render the launchd plist for this host. For --receiver the worker plist is DERIVED, not a second
560
666
  * template: same KeepAlive/ExitTimeOut shape, label and log names swapped, and the shared wrapper given
561
- * the receiver's exec argv instead of the worker's. The wrapper's exit-2 conversion is a no-op for the
562
- * receiver — it has no EXIT_POLICY — and harmless.
667
+ * the receiver's exec argv instead of the worker's. The wrapper's exit-2 conversion is LOAD-BEARING for
668
+ * the receiver too, not the no-op an earlier version of this comment claimed: since
669
+ * receiver/src/cli.mjs gained entryExitCode, a determinate receiver refusal exits EXIT_POLICY (2), and
670
+ * the wrapper's conversion to a clean 0 is exactly what KeepAlive/SuccessfulExit=false reads as
671
+ * "leave it stopped" -- launchd's only way to spell RestartPreventExitStatus=2.
563
672
  */
564
673
  function renderPlist(ctx) {
565
674
  // Substituted before anything composed goes in (subDeployDir): the two anchors below are therefore
@@ -650,7 +759,151 @@ function nssmSequence(ctx) {
650
759
  };
651
760
  }
652
761
 
653
- function doRender(ctx) {
762
+ /**
763
+ * Does this worker deployment run the podman venue, per the `.env` its unit loads? `{ used: false }`, `{ error }`,
764
+ * or `{ used: true, venues, env }`. Only the worker has a stack: the receiver runs no job and needs no Podman.
765
+ *
766
+ * The keys come from the deployment's `.env`, the file the unit's `EnvironmentFile=` loads, read the way that loader
767
+ * reads it (`readStackKeys`, shared with `up`). NOT this shell's environment: nothing in this project loads `.env` into
768
+ * a process (docs/secrets.md), so a shell that exports PI_BACKENDS=podman says nothing about what the SERVICE will
769
+ * run, and installing Quadlet units off it would stand up a stack for a worker that then runs docker. A key an
770
+ * `--env-setup` script exports is invisible here for the same reason, which docs/podman.md says.
771
+ *
772
+ * On macOS and Windows the answer decides only a NOTE (the venue runs only on Linux, so there is no stack to
773
+ * install), so it is read with that platform's own loader (the wrapper sources the file with sh; the cmd wrapper
774
+ * splits it) and a file it cannot read there is not a refusal: refusing an install over a key that could change
775
+ * nothing about it would be a refusal with no remedy.
776
+ */
777
+ function podmanVenue(ctx) {
778
+ if (ctx.which !== "worker") return { used: false };
779
+ const envPath = join(ctx.deployDir, ".env");
780
+ const linux = ctx.platform === "linux";
781
+ let fileKeys = {};
782
+ if (ctx.fs.existsSync(envPath)) {
783
+ let text;
784
+ try {
785
+ // Bytes, not text: `readStackKeys` checks what systemd refuses to load before it decodes (issue #447).
786
+ text = ctx.fs.readFileSync(envPath);
787
+ } catch (err) {
788
+ if (!linux) return { used: false };
789
+ return { error: `cannot read ${envPath} to learn whether this deployment runs the podman venue: ${err?.message}` };
790
+ }
791
+ const loader = linux ? "systemd" : ctx.platform === "darwin" ? "shell" : "cmd";
792
+ const read = readStackKeys(text, { loader, path: envPath });
793
+ if (read.error) return linux ? { error: read.error } : { used: false };
794
+ fileKeys = read.keys;
795
+ // Issue #464: where the worker's queue is, and whether a Valkey another uid holds may be it (PI_VALKEY_SHARED), as
796
+ // the service reads them, for the Valkey rule (`decideValkey`). A line the loaders read differently is refused,
797
+ // naming what in it is the problem, rather than guessed at.
798
+ const valkey = readValkeyKeys(text, { loader, path: envPath });
799
+ if (valkey.error) return linux ? { error: valkey.error } : { used: false };
800
+ fileKeys = { ...fileKeys, ...valkey.keys };
801
+ }
802
+ // Refused here, unlike doctor's venuesOf, which reads an unparseable list as `local`: doctor then reports the parse,
803
+ // but an install that guessed would write a unit and a stack for a deployment the worker refuses to boot.
804
+ try {
805
+ parseBackendList(fileKeys.PI_BACKENDS);
806
+ } catch (err) {
807
+ return linux ? { error: `${envPath}: ${err.message}` } : { used: false };
808
+ }
809
+ const venues = venuesOf(fileKeys);
810
+ if (!venues.podmanUsed) return { used: false };
811
+ return { used: true, venues, env: fileKeys };
812
+ }
813
+
814
+ /**
815
+ * The refusal for the podman venue in SYSTEM scope. Not "podman cannot run under a system unit": a system unit with
816
+ * `User=` runs rootless Podman fine (measured, docs/podman.md step 2). The stack is the problem. Its Quadlet units
817
+ * belong to the account's own user manager, and a system unit cannot `Wants=`/`After=` a unit of another manager, so
818
+ * the worker would race its own queue at every boot; and installing them means writing into a user's config from a
819
+ * command whose system-scope doctrine is to write nothing. User scope with linger does both properly.
820
+ */
821
+ function refusePodmanSystemScope(ctx) {
822
+ return fail(
823
+ ctx.err,
824
+ `PI_BACKENDS in ${join(ctx.deployDir, ".env")} lists podman, and the podman venue's stack (Valkey, the egress proxy) runs as Quadlet units in the worker account's OWN user manager, which a system unit cannot order itself after. Install in user scope (drop --system) with linger on: sudo loginctl enable-linger ${ctx.user}\nOr keep a hand-written system unit and start the stack by hand (docs/podman.md, steps 6 and 7).`,
825
+ );
826
+ }
827
+
828
+ /**
829
+ * The podman venue's stack for this install: which parts, rendered, with the actions that install them. `{ error }`
830
+ * for what cannot be installed, else `{ components, plan, notes }`. The Valkey rule differs from `up`'s on purpose:
831
+ * `up` starts what is not running, while an install also keeps (and orders the worker after) a Valkey unit an
832
+ * earlier run already installed, even though that Valkey is now the thing listening.
833
+ */
834
+ async function podmanStackFor(ctx, venue, { readKeeper = false } = {}) {
835
+ let armed;
836
+ try {
837
+ armed = egressArmed(venue.env);
838
+ } catch (err) {
839
+ return { error: `${join(ctx.deployDir, ".env")}: ${err.message}` };
840
+ }
841
+ const installed = ctx.fs.existsSync(join(quadletDir(ctx.home), QUADLET_FILES.valkey.file));
842
+ // Issue #464: a listener on any address VALKEY_URL reaches is taken to be this deployment's Valkey only when it is
843
+ // this account's (its uid, or a subordinate uid its containers run as), or PI_VALKEY_SHARED=1 in .env says it is
844
+ // shared on purpose; anything else is refused, not forceably (`stackRefusal`).
845
+ const valkey = await decideValkey({
846
+ venues: venue.venues,
847
+ url: venue.env.VALKEY_URL,
848
+ installed,
849
+ probeTcp: ctx.probeTcp,
850
+ lookup: ctx.lookup,
851
+ interfaces: ctx.interfaces,
852
+ euid: ctx.euid,
853
+ user: ctx.user,
854
+ shared: valkeySharedOn(venue.env[VALKEY_SHARED_KEY]),
855
+ subuids: readSubuidRanges({ user: ctx.user, euid: ctx.euid, fs: ctx.fs, run: ctx.runSync }),
856
+ fs: ctx.fs,
857
+ ownerName: (uid) => passwdNameFrom(ctx.fs, uid),
858
+ envPath: join(ctx.deployDir, ".env"),
859
+ });
860
+ if (valkey.error) return { error: valkey.error };
861
+ // Issue #468: the password the Quadlet Valkey starts with, when this install includes it: the deployment's own
862
+ // VALKEY_PASSWORD, or a new one this install writes into .env (never over a value), unless the Valkey is shared or
863
+ // the operator's own (`valkeyPasswordDecision`). Generated here, written by doInstall once nothing refuses; `render`
864
+ // writes nothing and shows the password file by its name alone.
865
+ let valkeyPassword = null;
866
+ let pendingPassword = null;
867
+ const passwordNotes = [];
868
+ if (valkey.include) {
869
+ const decided = valkeyPasswordDecision(venue.env, { envPath: join(ctx.deployDir, ".env") });
870
+ if (decided.error) return { error: decided.error };
871
+ if (decided.note) passwordNotes.push(decided.note);
872
+ if (decided.generate) pendingPassword = ctx.newPassword();
873
+ valkeyPassword = decided.password ?? pendingPassword;
874
+ }
875
+ const components = stackComponents({ venues: venue.venues, env: venue.env, includeValkey: valkey.include, armed, valkeyPort: valkey.port, valkeyPassword });
876
+ components.notes.push(...valkey.notes, ...passwordNotes);
877
+ // Gate round 2: the service reads PI_VALKEY_SHARED from .env alone, so a shell export says nothing; named, not honoured.
878
+ if (typeof ctx.env[VALKEY_SHARED_KEY] === "string") components.notes.push(sharedShellIgnored(ctx.env[VALKEY_SHARED_KEY], join(ctx.deployDir, ".env")));
879
+ // The keeper, read as `up` reads it (PR #463 final review), install only (render spawns nothing): a Quadlet keeper
880
+ // whose file is already there and unchanged but that does not hold (stopped, paused, off its bridge) is RESTARTED,
881
+ // since `start` over it is a no-op for a paused one and would say nothing of the proxy either way; the plan then
882
+ // restarts our proxy after it (planStack), or the install names the operator's own proxy to restart.
883
+ let restartUnits = [];
884
+ if (readKeeper && components.keeper && ctx.fs.existsSync(join(quadletDir(ctx.home), QUADLET_FILES.keeper.file))) {
885
+ const keeper = judgeNetnsKeeper(await runQuery(ctx, "podman", ["inspect", NETNS_KEEPER_FORMAT, NETNS_KEEPER]));
886
+ if (!keeper.holds) restartUnits = [QUADLET_FILES.keeper.unit];
887
+ }
888
+ let plan = planStack({ components, templatesDir: ctx.templatesDir, deployDir: ctx.deployDir, home: ctx.home, fs: ctx.fs, restartUnits });
889
+ if (plan.error) return { error: plan.error };
890
+ // Issue #468: a Valkey that already ran and now gets a new password (an upgrade from a deployment that had none) is one
891
+ // its running clients can no longer talk to, so the worker and the receiver units this account runs are restarted
892
+ // after it. Asked only then (install only; render spawns nothing), and only of units the manager says are active.
893
+ const secret = plan.files.find((f) => f.kind === "secret");
894
+ const valkeyFile = plan.files.find((f) => f.unit === QUADLET_FILES.valkey.unit && f.path.endsWith(".container"));
895
+ if (readKeeper && secret && secret.state !== "same" && valkeyFile && valkeyFile.state !== "new") {
896
+ const active = [];
897
+ for (const unit of ["pi-dispatch-worker.service", "pi-dispatch-receiver.service"]) {
898
+ const res = await runQuery(ctx, "systemctl", ["--user", "is-active", unit]);
899
+ if (res.code === 0 && String(res.stdout ?? "").trim() === "active") active.push(unit);
900
+ }
901
+ if (active.length > 0) plan = planStack({ components, templatesDir: ctx.templatesDir, deployDir: ctx.deployDir, home: ctx.home, fs: ctx.fs, restartUnits, restartAfterValkey: active });
902
+ }
903
+ return { components, plan, notes: components.notes, venues: venue.venues, proxy: egressProxyName(venue.env), valkeyRefusal: valkey.refusal, pendingPassword };
904
+ }
905
+
906
+ async function doRender(ctx) {
654
907
  if (ctx.which === "receiver" && !ctx.receiverStart) return refuseMissingReceiver(ctx);
655
908
  const paths = unitPaths(ctx);
656
909
  if (ctx.platform === "darwin") {
@@ -667,8 +920,26 @@ function doRender(ctx) {
667
920
  return 0;
668
921
  }
669
922
  if (ctx.platform === "linux") {
923
+ const venue = podmanVenue(ctx);
924
+ if (venue.error) return fail(ctx.err, venue.error);
925
+ let stack = null;
926
+ if (venue.used) {
927
+ if (ctx.scope === "system") return refusePodmanSystemScope(ctx);
928
+ stack = await podmanStackFor(ctx, venue);
929
+ if (stack.error) return fail(ctx.err, stack.error);
930
+ }
670
931
  ctx.out(`# → ${paths.installPath}\n`);
671
- ctx.out(renderLinuxUnit(ctx));
932
+ ctx.out(renderLinuxUnit(ctx, stack?.plan.start ?? [], stack?.venues ?? null));
933
+ // The proxy's rules copy is named, not printed (round 3 nit): it is the package's egress-proxy.conf verbatim, and
934
+ // 4.7 KB of squid configuration under a "Quadlet" heading read as a unit file.
935
+ for (const f of stack?.plan.files ?? []) {
936
+ if (f.kind === "conf") ctx.out(`\n# → ${f.path} (the egress proxy's rules: install copies the package's egress-proxy.conf here, unchanged)\n`);
937
+ // Issue #468: the password file is NAMED, never printed: render output lands in scrollbacks and bug reports.
938
+ else if (f.kind === "secret") ctx.out(`\n# → ${f.path} (mode 0600: ${VALKEY_PASSWORD_KEY} for the Quadlet Valkey, ${f.text.includes(`${VALKEY_PASSWORD_KEY}=\n`) ? "empty, so it starts without a password" : "from .env or generated into it at install; the value is not shown"})\n`);
939
+ else ctx.out(`\n# → ${f.path} (Quadlet, podman venue)\n${f.text}`);
940
+ }
941
+ for (const note of stack?.notes ?? []) ctx.out(`# note: ${note}\n`);
942
+ if (stack?.valkeyRefusal) ctx.out(`# note: install refuses this: ${stack.valkeyRefusal.text}\n`);
672
943
  return 0;
673
944
  }
674
945
  const { service, commands } = nssmSequence(ctx);
@@ -697,9 +968,27 @@ async function doInstall(ctx) {
697
968
  }
698
969
  }
699
970
 
700
- if (ctx.platform === "darwin") return installDarwin(ctx, paths);
701
- if (ctx.platform === "linux") return ctx.scope === "system" ? installLinuxSystem(ctx, paths) : installLinuxUser(ctx, paths);
702
- return installWindows(ctx);
971
+ const venue = podmanVenue(ctx);
972
+ if (venue.error) return fail(ctx.err, venue.error);
973
+ if (ctx.platform === "darwin" || ctx.platform === "win32") {
974
+ const code = ctx.platform === "darwin" ? await installDarwin(ctx, paths) : await installWindows(ctx);
975
+ // Said, never silently skipped: the venue refuses every host that is not Linux (podman-platform), so there is
976
+ // no stack to install here, and a deployment listing it should know why nothing about Podman happened.
977
+ if (code === 0 && venue.used) ctx.out("note: PI_BACKENDS lists podman, and the podman venue runs only on Linux (it refuses this host as podman-platform), so no podman stack was installed here\n");
978
+ return code;
979
+ }
980
+ if (ctx.scope === "system") return venue.used ? refusePodmanSystemScope(ctx) : installLinuxSystem(ctx, paths);
981
+ let stack = null;
982
+ let foreign = [];
983
+ if (venue.used) {
984
+ stack = await podmanStackFor(ctx, venue, { readKeeper: true });
985
+ if (stack.error) return fail(ctx.err, stack.error);
986
+ // Every stack refusal first, the unit's own among them, so installLinuxUser's unit check never fires alone.
987
+ const refused = await stackRefusal(ctx, paths, stack);
988
+ if (refused.error) return fail(ctx.err, refused.error);
989
+ foreign = refused.foreign;
990
+ }
991
+ return installLinuxUser(ctx, paths, stack, foreign);
703
992
  }
704
993
 
705
994
  async function installDarwin(ctx, paths) {
@@ -721,11 +1010,18 @@ async function installDarwin(ctx, paths) {
721
1010
  // the old copy out first. "Not loaded" is a fine answer — the nonzero exit is ignored.
722
1011
  await run(ctx, "launchctl", ["bootout", `gui/${ctx.euid}/${paths.name}`]);
723
1012
  }
724
- ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
725
- // launchd creates the StandardOutPath FILES but not their parent directory: without this the job
726
- // spawns and dies with its error unwritable. Done at install, not render — render stays read-only.
727
- ctx.fs.mkdirSync(join(ctx.deployDir, "logs"), { recursive: true });
728
- ctx.fs.writeFileSync(paths.installPath, renderPlist(ctx));
1013
+ // Issue #464: a write that fails is said, with what this run left, rather than thrown out of the command. The plist
1014
+ // is the one file this path writes, so a failure leaves none of ours (a failed write is put back like the Linux path's).
1015
+ const journal = [];
1016
+ try {
1017
+ ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
1018
+ // launchd creates the StandardOutPath FILES but not their parent directory: without this the job
1019
+ // spawns and dies with its error unwritable. Done at install, not render: render stays read-only.
1020
+ ctx.fs.mkdirSync(join(ctx.deployDir, "logs"), { recursive: true });
1021
+ journalWrite(ctx.fs, journal, paths.installPath, renderPlist(ctx));
1022
+ } catch (err) {
1023
+ return fail(ctx.err, `could not write ${paths.installPath} (${err?.message ?? err}), so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}`);
1024
+ }
729
1025
  const bootstrap = await run(ctx, "launchctl", ["bootstrap", `gui/${ctx.euid}`, paths.installPath]);
730
1026
  if (bootstrap !== 0) {
731
1027
  return fail(ctx.err, `launchctl bootstrap failed (exit ${bootstrap}) — the plist is written; retry by hand: launchctl bootstrap gui/${ctx.euid} ${paths.installPath}`);
@@ -740,25 +1036,176 @@ async function installDarwin(ctx, paths) {
740
1036
  return 0;
741
1037
  }
742
1038
 
743
- async function installLinuxUser(ctx, paths) {
1039
+ /**
1040
+ * Every reason the podman venue's stack refuses an install, gathered BEFORE anything is written (round 2, E9): the
1041
+ * worker unit existing used to be the only thing said, so an operator re-ran with --force and then met a replaced
1042
+ * container and a restarted proxy nobody had mentioned. `--force` is consent to exactly the list printed here. Three
1043
+ * reasons are not forceable: a missing allowlist, a container whose state could not be read, and no user manager.
1044
+ * Returns `{ error }` or `{ foreign }` (the foreign containers --force will replace, for the warnings).
1045
+ */
1046
+ async function stackRefusal(ctx, paths, stack) {
1047
+ const blocking = [];
1048
+ const forceable = [];
1049
+ if (ctx.fs.existsSync(paths.installPath)) forceable.push(`${paths.installPath} already exists (same non-clobber contract as init); --force replaces it`);
1050
+ const changed = stack.plan.files.filter((f) => f.state === "changed");
1051
+ if (changed.length > 0) forceable.push(`${changed.map((f) => f.path).join(", ")} already ${changed.length === 1 ? "exists" : "exist"} with other content than this version renders; --force replaces ${changed.length === 1 ? "it" : "them"} and restarts ${(stack.plan.restart ?? []).join(" ") || "nothing"}${proxyRestartWarning(stack.plan) ? ` (${proxyRestartWarning(stack.plan)})` : ""}`);
1052
+ // Issue #464: a Valkey VALKEY_URL reaches that is not this account's. NOT forceable: --force is how an operator replaces
1053
+ // a changed file, and taking another account's queue must never ride along with that. The opt-in is its own named key.
1054
+ if (stack.valkeyRefusal) blocking.push(stack.valkeyRefusal.text);
1055
+ const allowlist = join(ctx.deployDir, "egress-allowlist.conf");
1056
+ if (stack.components.proxy && !ctx.fs.existsSync(allowlist)) {
1057
+ blocking.push(`the egress policy is on, and ${allowlist} does not exist: a proxy unit mounting a missing file makes Podman create a DIRECTORY there and squid fail confusingly. Run \`pi-dispatch init\` in ${ctx.deployDir} first (it never overwrites), or set PI_EGRESS=0 in .env to opt out of the policy`);
1058
+ }
1059
+ const bus = stack.plan.actions.length > 0 ? userBusRefusal({ env: ctx.env, user: ctx.user, euid: ctx.euid }) : null;
1060
+ if (bus) blocking.push(bus);
1061
+ // The manager's own XDG_RUNTIME_DIR and XDG_CONFIG_HOME (PR #463 round 3, measured): another account's makes every
1062
+ // unit this installs fail to start, or not exist at all. Asked only where the manager can be asked.
1063
+ if (!bus && stack.plan.actions.length > 0) {
1064
+ const managerEnv = managerEnvRefusal(await runQuery(ctx, "systemctl", ["--user", "show-environment"]), { home: ctx.home, euid: ctx.euid, user: ctx.user, realpath: ctx.realpath });
1065
+ if (managerEnv) blocking.push(managerEnv);
1066
+ }
1067
+ // A container of the unit's name that the unit does not own would be removed by its `podman run --replace`
1068
+ // (issue #430 review): refused unless --force says to replace it, and then said out loud.
1069
+ const containers = await foreignContainers(stack.plan, (cmd, args) => runQuery(ctx, cmd, args));
1070
+ const foreign = containers.found;
1071
+ if (containers.unknown.length > 0) blocking.push(unknownContainerRefusal(containers.unknown));
1072
+ if (foreign.length > 0) forceable.push(foreignContainerRefusal(foreign, { forceHint: "pass --force to let the Quadlet unit replace it" }));
1073
+ if (blocking.length > 0 || (forceable.length > 0 && !ctx.force)) {
1074
+ const lines = [...blocking, ...(ctx.force ? [] : forceable)];
1075
+ // Worded from the counts, since both lists can hold several (round 3 nit): the blocking reasons come first.
1076
+ const byHand = blocking.length === 1 ? "The first item must be fixed by hand" : `The first ${blocking.length} items must be fixed by hand`;
1077
+ const tail = blocking.length === 0 ? "\n--force accepts every item above at once." : forceable.length > 0 && !ctx.force ? `\n${byHand}; --force accepts the rest.` : `\n${blocking.length === 1 ? "It" : "Each"} must be fixed by hand; --force does not apply.`;
1078
+ return { error: `${lines.length === 1 ? lines[0] : `nothing installed, for these reasons:\n${lines.map((l) => ` - ${l}`).join("\n")}`}${lines.length > 1 ? tail : ""}` };
1079
+ }
1080
+ return { foreign };
1081
+ }
1082
+
1083
+ async function installLinuxUser(ctx, paths, stack = null, foreign = []) {
744
1084
  if (ctx.fs.existsSync(paths.installPath) && !ctx.force) {
745
1085
  return fail(ctx.err, `${paths.installPath} already exists — pass --force to replace it (same non-clobber contract as init)`);
746
1086
  }
747
- ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
748
- ctx.fs.writeFileSync(paths.installPath, renderLinuxUnit(ctx));
1087
+ // Issue #464: every file this run writes, recorded before it is written, so a failure puts them back rather than
1088
+ // leaving half an install (seen with a root-owned ~/.config: the stack's files written, the worker unit refused). All
1089
+ // files are written before any command runs (the worker unit inside applyStack's `beforeRuns`), which is what makes a
1090
+ // write failure all-or-nothing. A COMMAND that fails leaves the stack's files, which units already started from, and
1091
+ // says exactly which files this run wrote.
1092
+ const journal = [];
1093
+ const writeUnit = () => {
1094
+ try {
1095
+ ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
1096
+ journalWrite(ctx.fs, journal, paths.installPath, renderLinuxUnit(ctx, stack?.plan.start ?? [], stack?.venues ?? null));
1097
+ return { ok: true };
1098
+ } catch (err) {
1099
+ return { ok: false, failed: `write ${paths.installPath}`, code: null, message: err?.message };
1100
+ }
1101
+ };
1102
+ const writtenList = () => [...new Set(journal.map((e) => e.path))].join(", ") || "none";
1103
+ if (stack?.pendingPassword) {
1104
+ // Issue #468: the new Valkey password into .env first, journalled like every other file of this run, so a write
1105
+ // that fails later puts .env back too. Never over a value (the writer's never-clobber), the file narrowed to this
1106
+ // account alone, the value never shown.
1107
+ const wrote = writeValkeyPassword(ctx, stack.pendingPassword, journal);
1108
+ if (wrote.error) return fail(ctx.err, `${wrote.error}, so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}`);
1109
+ ctx.out(`generated ${VALKEY_PASSWORD_KEY} into ${wrote.path} (32 random bytes, hex; the value is not shown; the file is now readable by this account only)\n`);
1110
+ }
1111
+ if (stack) {
1112
+ for (const note of stack.notes) ctx.out(`note: ${note}\n`);
1113
+ for (const f of foreign) {
1114
+ ctx.out(`⚠ --force: ${f.container} is not managed by ${f.unit} and will be REPLACED by it${f.unit === QUADLET_FILES.proxy.unit ? "; every job running right now loses the per-job network it had on that proxy" : ""}\n`);
1115
+ }
1116
+ const restartWarning = proxyRestartWarning(stack.plan);
1117
+ if (restartWarning) ctx.out(`⚠ ${restartWarning}\n`);
1118
+ const passwordWarning = valkeyPasswordRestartWarning(stack.plan);
1119
+ if (passwordWarning) ctx.out(`⚠ ${passwordWarning}\n`);
1120
+ if (stack.plan.actions.length > 0) {
1121
+ // Shown, then done: the same lines `up` asks consent for, carried out by the same function.
1122
+ ctx.out(`podman venue (PI_BACKENDS in .env): its stack, as Quadlet units in ${stack.plan.dir}:\n`);
1123
+ for (const action of stack.plan.actions) ctx.out(` ${describeAction(action)}\n`);
1124
+ const applied = await applyStack(stack.plan, { fs: ctx.fs, run: (cmd, args) => run(ctx, cmd, args), journal, beforeRuns: writeUnit });
1125
+ if (!applied.ok) {
1126
+ const what = `${applied.failed} failed (${applied.code === null ? applied.message ?? "command not found" : `exit ${applied.code}`})`;
1127
+ if (!applied.ran) {
1128
+ return fail(ctx.err, `${what}, so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}. Fix it, then re-run this install`);
1129
+ }
1130
+ // A command failed: the stack's units may be running from their files, so only the worker unit is put back,
1131
+ // and the manager is reloaded so it forgets it (best effort; the unit was never enabled or started).
1132
+ const unitOnly = journal.filter((e) => e.path === paths.installPath);
1133
+ const rolled = rollBackWrites(ctx.fs, unitOnly);
1134
+ if (rolled.left.length === 0) await run(ctx, "systemctl", ["--user", "daemon-reload"]);
1135
+ const remain = [...new Set(journal.filter((e) => e.path !== paths.installPath).map((e) => e.path))];
1136
+ return fail(
1137
+ ctx.err,
1138
+ `${what}, so the worker unit was NOT installed: it would only start against a missing queue or proxy (${describeRollBack(rolled, { partial: true })}). The stack files this run wrote remain, since its units run from them: ${remain.join(", ") || "none"}. \`pi-dispatch service uninstall\` removes them if you will not retry. \`systemctl --user status ${stack.plan.start.join(" ")}\` and \`journalctl --user -u <unit>\` have the details; fix it, then re-run this install`,
1139
+ );
1140
+ }
1141
+ }
1142
+ }
1143
+ if (!journal.some((e) => e.path === paths.installPath)) {
1144
+ const wrote = writeUnit();
1145
+ if (!wrote.ok) return fail(ctx.err, `${wrote.failed} failed (${wrote.message}), so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}. Fix it, then re-run this install`);
1146
+ }
749
1147
  const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
750
- if (reload === null) return fail(ctx.err, `systemctl not found — is this a systemd host? The unit is written at ${paths.installPath}`);
1148
+ if (reload === null) return fail(ctx.err, `systemctl not found: is this a systemd host? The unit is written at ${paths.installPath} (files this run wrote, which remain: ${writtenList()})`);
751
1149
  const enable = await run(ctx, "systemctl", ["--user", "enable", "--now", paths.name]);
752
1150
  if (enable !== 0) {
753
- return fail(ctx.err, `systemctl --user enable --now ${paths.name} failed (exit ${enable}) — the unit is written at ${paths.installPath}; \`systemctl --user status ${paths.name}\` has the details`);
1151
+ return fail(ctx.err, `systemctl --user enable --now ${paths.name} failed (exit ${enable}); the unit is written at ${paths.installPath}; \`systemctl --user status ${paths.name}\` has the details (files this run wrote, which remain: ${writtenList()})`);
754
1152
  }
755
1153
  ctx.out(`installed ${paths.name} → ${paths.installPath} (enabled and started in your user manager)\n`);
1154
+ if (stack) {
1155
+ // Said as it happened: `start` on a unit already active is a no-op, so a unit whose file this run replaced was
1156
+ // RESTARTED (planStack), and the line names which is which rather than calling everything "started".
1157
+ const restarted = stack.plan.restart ?? [];
1158
+ const started = stack.plan.start.filter((u) => !restarted.includes(u));
1159
+ if (started.length > 0) ctx.out(`started ${started.join(" ")} (Quadlet units: never enabled, the generator reads their own [Install] section; a unit already running is left as it is)\n`);
1160
+ // Named by the file that changed (round 3 nit): the proxy also restarts for its rules copy alone.
1161
+ for (const unit of restarted) {
1162
+ // Issue #468: a file this install wrote NEW for a unit it restarts (the Valkey's first password file) is named too.
1163
+ const changedFiles = stack.plan.files.filter((f) => f.state !== "same" && f.restarts === unit).map((f) => f.path);
1164
+ // A unit restarted with its files unchanged (a keeper that did not hold, the proxy after it) says so rather than
1165
+ // ending on "replaced " with nothing after it, as it did.
1166
+ ctx.out(`restarted ${unit}, ${changedFiles.length > 0 ? `because this install replaced ${changedFiles.join(" and ")}` : "as the plan above says (its files are unchanged)"}\n`);
1167
+ }
1168
+ if ((stack.plan.clientsRestarted ?? []).length > 0) ctx.out(`restarted ${stack.plan.clientsRestarted.join(" ")} after it, so they send the new ${VALKEY_PASSWORD_KEY}\n`);
1169
+ if (stack.plan.start.length > 0) ctx.out(`${paths.name} Wants= and is After= ${stack.plan.start.join(" ")}\n`);
1170
+ // PR #463 round 3: a keeper started or restarted beside a proxy this install does not own (PI_EGRESS_PROXY naming
1171
+ // the operator's own) leaves that proxy up since before the keeper; said, with its name.
1172
+ const hint = keeperUnderRunningProxyHint(stack.plan, stack.proxy);
1173
+ if (hint) ctx.out(`⚠ ${hint}\n`);
1174
+ // A WARNING, not a refusal, like the note below it: a worker that runs only while its operator is logged in is a
1175
+ // real (desktop) deployment. But on this venue linger is also what brings the queue and the proxy back, and it
1176
+ // was measured both ways, so the line says which of the two this host is rather than the general caution.
1177
+ ctx.out(lingerNote(await readLinger(ctx.user, (cmd, args) => runCapture(ctx, cmd, args)), ctx.user));
1178
+ return 0;
1179
+ }
756
1180
  // Without linger a user manager only runs while a session exists — fine on a desktop, a silent
757
1181
  // no-worker-after-reboot on a headless box. Say so instead of letting the operator find out.
758
1182
  ctx.out(`note: user units run while you have a session. For a headless host that must start at boot: sudo loginctl enable-linger ${ctx.user}\n`);
759
1183
  return 0;
760
1184
  }
761
1185
 
1186
+ /**
1187
+ * Write a new VALKEY_PASSWORD into the deployment's `.env` (issue #468) through the project's one `.env` writer, never
1188
+ * over a value, the file narrowed to this account alone (`narrow`). Recorded in `journal` with the bytes that were there,
1189
+ * so a later failure of the same install puts it back. `{ path }` or `{ error }`; the value is never in either.
1190
+ */
1191
+ function writeValkeyPassword(ctx, password, journal) {
1192
+ const envPath = join(ctx.deployDir, ".env");
1193
+ let previous;
1194
+ try {
1195
+ previous = ctx.fs.readFileSync(envPath);
1196
+ } catch (err) {
1197
+ return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: it could not be read (${err?.message ?? err})` };
1198
+ }
1199
+ try {
1200
+ const res = updateEnvFile(envPath, VALKEY_PASSWORD_KEY, password, { fs: ctx.fs, platform: ctx.platform, narrow: true });
1201
+ if (!res.changed) return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: the file already assigns it (set while this install ran?). Re-run the install, which then uses that value` };
1202
+ } catch (err) {
1203
+ return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: ${err?.message ?? err}` };
1204
+ }
1205
+ journal.push({ path: envPath, existed: true, previous });
1206
+ return { path: envPath };
1207
+ }
1208
+
762
1209
  async function installLinuxSystem(ctx, paths) {
763
1210
  if (ctx.fs.existsSync(paths.installPath) && !ctx.force) {
764
1211
  return fail(ctx.err, `${paths.installPath} already exists — pass --force to re-stage the render (the printed sudo commands would overwrite it)`);
@@ -815,6 +1262,20 @@ async function doUninstall(ctx) {
815
1262
  return 0;
816
1263
  }
817
1264
  const userPath = ctx.platform === "darwin" ? paths.installPath : paths.userPath;
1265
+ // The podman venue's Quadlet units (issue #430), removed with the worker, or on their own when `up` installed them
1266
+ // and no worker unit was ever written. Decided by what EXISTS, not by today's PI_BACKENDS: an operator who dropped
1267
+ // podman from the list still has the units, and they are still this tool's to remove.
1268
+ const quadlets = ctx.platform === "linux" && ctx.which === "worker" && ctx.scope === "user" ? ALL_QUADLET_FILES.filter((q) => ctx.fs.existsSync(join(quadletDir(ctx.home), q.file))) : [];
1269
+ // No user manager to talk to (`sudo -iu`, round 3 D3, measured): every `systemctl --user` below would fail, and the
1270
+ // old code ignored the exits, printed success with exit 0, and left the worker and the stack running with a
1271
+ // dangling default.target.wants link. Refused before anything is touched, whenever this run would talk to it.
1272
+ if (ctx.platform === "linux" && (ctx.fs.existsSync(userPath) || quadlets.length > 0)) {
1273
+ const bus = userBusRefusal({ env: ctx.env, user: ctx.user, euid: ctx.euid });
1274
+ if (bus) return fail(ctx.err, bus.replace("so `systemctl --user` would fail after the files were written", "so `systemctl --user` could stop and disable nothing"));
1275
+ }
1276
+ if (!ctx.fs.existsSync(userPath) && quadlets.length > 0 && !ctx.fs.existsSync(paths.systemPath)) {
1277
+ return removeQuadlets(ctx, quadlets);
1278
+ }
818
1279
  if (!ctx.fs.existsSync(userPath)) {
819
1280
  // Say where it looked — both scopes — and if the unit turns out to live in ROOT scope, print
820
1281
  // the removal commands instead of touching them (the same never-root doctrine as install).
@@ -833,13 +1294,94 @@ async function doUninstall(ctx) {
833
1294
  ctx.out(`uninstalled ${paths.name} (booted out of gui/${ctx.euid}, plist removed)\n`);
834
1295
  return 0;
835
1296
  }
836
- await run(ctx, "systemctl", ["--user", "disable", "--now", paths.name]); // not-enabled is fine
1297
+ // A non-zero exit here is a unit that is still enabled or still running (round 3, D3): nothing is removed and nothing
1298
+ // is claimed. `disable --now` on an installed unit that is merely not enabled exits 0.
1299
+ const disabled = await run(ctx, "systemctl", ["--user", "disable", "--now", paths.name]);
1300
+ if (disabled !== 0) {
1301
+ return fail(ctx.err, `systemctl --user disable --now ${paths.name} failed (${disabled === null ? "systemctl not found" : `exit ${disabled}`}), so nothing was removed: the worker may still be running. \`systemctl --user status ${paths.name}\` has the details`);
1302
+ }
837
1303
  ctx.fs.unlinkSync(userPath);
838
- await run(ctx, "systemctl", ["--user", "daemon-reload"]);
1304
+ const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
1305
+ if (reload !== 0) {
1306
+ return fail(ctx.err, `${paths.name} is disabled, stopped and its unit file removed, but systemctl --user daemon-reload failed (${reload === null ? "systemctl not found" : `exit ${reload}`}): run it yourself`);
1307
+ }
839
1308
  ctx.out(`uninstalled ${paths.name} (disabled, stopped, unit removed)\n`);
1309
+ if (quadlets.length > 0) return removeQuadlets(ctx, quadlets);
840
1310
  return 0;
841
1311
  }
842
1312
 
1313
+ /**
1314
+ * Stop and remove the podman venue's Quadlet units. `stop`, never `disable`: a generated unit cannot be disabled any
1315
+ * more than enabled, and removing its file plus a daemon-reload is what unlinks it from default.target. The Valkey
1316
+ * volume and the three networks are deliberately LEFT: the volume holds the queue's wait-list, and a re-install that
1317
+ * found it gone would have dropped every waiting job on an uninstall nobody meant as a purge.
1318
+ */
1319
+ async function removeQuadlets(ctx, quadlets) {
1320
+ // The network units too (round 3, D2, measured): left behind, they stay `active (exited)` in the manager, so after the
1321
+ // `podman network rm` the message below suggests, a reinstall in the same manager lifetime found its network unit
1322
+ // already "started" and the container failed with "network not found". Containers first, then their networks.
1323
+ const containers = quadlets.filter((q) => q.file.endsWith(".container")).map((q) => q.unit);
1324
+ const networks = quadlets.filter((q) => q.file.endsWith(".network")).map((q) => q.unit);
1325
+ const units = [...containers, ...networks];
1326
+ if (units.length > 0) {
1327
+ // A unit that is already stopped stops with exit 0; a non-zero exit is a stop that did not happen (no manager, a
1328
+ // unit that would not die), and removing the files under a running container would leave it running unmanaged
1329
+ // while this command claimed it was gone (round 3, D3).
1330
+ const stopped = await run(ctx, "systemctl", ["--user", "stop", ...units]);
1331
+ // Issue #464: a unit the manager never loaded (an install or `up` whose daemon-reload failed after writing the
1332
+ // files) makes `stop` exit 5 ("not loaded"), and uninstall then could not remove the very files that failed run
1333
+ // left. So a failed stop is asked about unit by unit: one that is not loaded, or loaded and not running, is
1334
+ // stopped as far as it can be; only a unit still active refuses the removal, as before.
1335
+ if (stopped !== 0 && stopped !== null && (await unitsStillRunning(ctx, units)).length === 0) {
1336
+ ctx.out(`note: systemctl --user stop exited ${stopped}, and none of ${units.join(", ")} is running (not loaded, or already stopped), so their files are removed\n`);
1337
+ } else if (stopped !== 0) {
1338
+ return fail(ctx.err, `systemctl --user stop ${units.join(" ")} failed (${stopped === null ? "systemctl not found" : `exit ${stopped}`}), so nothing was removed: the containers may still be running. \`systemctl --user status ${units.join(" ")}\` has the details`);
1339
+ }
1340
+ }
1341
+ for (const q of quadlets) ctx.fs.unlinkSync(join(quadletDir(ctx.home), q.file));
1342
+ // The account-owned copy of the proxy's rules (E1) goes with its unit, and the Valkey's password file with its unit
1343
+ // (issue #468; the password itself stays in the deployment's .env, where a re-install finds it).
1344
+ const conf = proxyConfCopyPath(ctx.home);
1345
+ if (ctx.fs.existsSync(conf)) ctx.fs.unlinkSync(conf);
1346
+ const valkeyEnv = valkeyEnvPath(ctx.home);
1347
+ if (quadlets.some((q) => q.file === QUADLET_FILES.valkey.file) && ctx.fs.existsSync(valkeyEnv)) ctx.fs.unlinkSync(valkeyEnv);
1348
+ const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
1349
+ if (reload !== 0) {
1350
+ return fail(ctx.err, `the Quadlet files are removed and their units stopped, but systemctl --user daemon-reload failed (${reload === null ? "systemctl not found" : `exit ${reload}`}): run it yourself so the manager forgets them`);
1351
+ }
1352
+ // squid ignores SIGTERM, so its stop takes systemd's 10 s and ends in SIGKILL, exit 137, and the unit is left
1353
+ // `failed` (measured): after the file is gone `systemctl --user list-units` would show a not-found failed unit
1354
+ // forever. reset-failed clears it. Its exit is not checked: a unit that never failed is already what it asks for.
1355
+ if (units.length > 0) await run(ctx, "systemctl", ["--user", "reset-failed", ...units]);
1356
+ // The volume is named only when a Valkey unit was among what was removed (issue #452 gate round 2): a stack installed
1357
+ // without Valkey never had one, and the line used to claim a volume nothing here made was kept.
1358
+ const hadValkey = quadlets.some((q) => q.file === QUADLET_FILES.valkey.file);
1359
+ ctx.out(`removed the podman venue's Quadlet units (${quadlets.map((q) => q.file).join(", ")}); ${hadValkey ? "the pi-dispatch-valkey-data volume and the networks are" : "the networks are"} kept, remove them with podman if you mean to\n`);
1360
+ return 0;
1361
+ }
1362
+
1363
+ /**
1364
+ * The units of `units` the user manager still has running, asked one by one (`systemctl --user show`), for an
1365
+ * uninstall whose `stop` failed (issue #464). A unit is taken as not running only on the manager's own answer: not
1366
+ * loaded (`LoadState=not-found`), or loaded and `inactive` or `failed`. A unit it could not be asked about counts as
1367
+ * running, so the removal is refused rather than guessed.
1368
+ */
1369
+ async function unitsStillRunning(ctx, units) {
1370
+ const running = [];
1371
+ for (const unit of units) {
1372
+ const shown = await runQuery(ctx, "systemctl", ["--user", "show", "--property=LoadState,ActiveState", unit]);
1373
+ const props = Object.fromEntries(
1374
+ String(shown.stdout ?? "")
1375
+ .split("\n")
1376
+ .map((l) => l.trim().split("="))
1377
+ .filter((kv) => kv.length === 2),
1378
+ );
1379
+ const idle = shown.code === 0 && (props.LoadState === "not-found" || (props.LoadState === "loaded" && ["inactive", "failed"].includes(props.ActiveState)));
1380
+ if (!idle) running.push(unit);
1381
+ }
1382
+ return running;
1383
+ }
1384
+
843
1385
  /** Informational only — reports every scope it knows about and always exits 0. */
844
1386
  async function doStatus(ctx) {
845
1387
  const paths = unitPaths(ctx);
@@ -875,6 +1417,16 @@ async function doStatus(ctx) {
875
1417
  const systemHits = [paths.systemPath, ...(ctx.which === "worker" ? ["/etc/systemd/system/worker.service"] : [])].filter((p) => ctx.fs.existsSync(p));
876
1418
  ctx.out(systemHits.length ? `system scope: ${systemHits.join(", ")} EXISTS — not managed by this tool\n` : "system scope: none\n");
877
1419
  reportEnvSetup(ctx, [paths.userPath, ...systemHits]);
1420
+ // The podman venue's Quadlet units, one line each, and nothing at all where there are none, so the output of every
1421
+ // other deployment is byte-identical to before issue #430.
1422
+ if (ctx.which === "worker") {
1423
+ for (const q of ALL_QUADLET_FILES) {
1424
+ const path = join(quadletDir(ctx.home), q.file);
1425
+ if (!ctx.fs.existsSync(path)) continue;
1426
+ const active = await runCapture(ctx, "systemctl", ["--user", "is-active", q.unit]);
1427
+ ctx.out(`quadlet: ${path} (${q.unit}): ${active.code === null ? "systemctl not found" : active.output.trim() || "unknown"}\n`);
1428
+ }
1429
+ }
878
1430
  return 0;
879
1431
  }
880
1432
 
@@ -945,13 +1497,15 @@ async function doRestart(ctx, values) {
945
1497
  return fail(ctx.err, `--drain-timeout must be a positive number of seconds, got: ${values["drain-timeout"]}`);
946
1498
  }
947
1499
  let queue = ctx.queue;
1500
+ let url;
948
1501
  if (!queue) {
949
1502
  // VALKEY_URL only, exactly like cli.mjs's pause/resume: the drain must work even when the rest
950
1503
  // of the config (forge auth …) is broken, and failFast keeps a down Valkey an error in seconds
951
1504
  // instead of a hung restart. Lazy imports for the same reason cli.mjs uses them: `service`
952
1505
  // subcommands that never touch the queue must not load bullmq/ioredis.
953
- const url = ctx.env.VALKEY_URL ?? "redis://127.0.0.1:6379";
954
- const { parseConnection } = await import("./connection.mjs");
1506
+ // This shell's VALKEY_URL, else the deployment .env's (PR #475's review), as every CLI verb reads it.
1507
+ const { cliValkeyUrl, parseConnection } = await import("./connection.mjs");
1508
+ url = cliValkeyUrl(ctx.env, { cwd: ctx.deployDir, warn: (line) => ctx.err(line) });
955
1509
  const { makeQueue } = await import("./queue.mjs");
956
1510
  queue = makeQueue(parseConnection(url, { failFast: true }));
957
1511
  }
@@ -981,7 +1535,7 @@ async function doRestart(ctx, values) {
981
1535
  // this command would otherwise be doing: reporting a drained queue it cannot see all of.
982
1536
  const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
983
1537
  if (delayed > 0) {
984
- ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
1538
+ ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, scope/host deferrals, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
985
1539
  }
986
1540
  const stopped = await doStop(ctx);
987
1541
  if (stopped !== 0) {
@@ -997,7 +1551,7 @@ async function doRestart(ctx, values) {
997
1551
  ctx.out("resumed — drained restart complete\n");
998
1552
  return 0;
999
1553
  } catch (error) {
1000
- return fail(ctx.err, `could not reach Valkey — is it running? (docker compose up)\n ${error.message}`);
1554
+ return fail(ctx.err, `could not reach Valkey: ${(await import("./valkey-auth.mjs")).valkeyDownHint(url)}\n ${error.message}`);
1001
1555
  } finally {
1002
1556
  await queue.close().catch(() => {});
1003
1557
  }
@@ -1013,6 +1567,20 @@ function quoteArgs(args) {
1013
1567
  return args.map((a) => (a.includes(" ") || a.includes("\\") ? `"${a}"` : a)).join(" ");
1014
1568
  }
1015
1569
 
1570
+ /** Is anything listening on host:port? up.mjs's probe, for up.mjs's reason: a plain connect, no protocol. */
1571
+ function defaultProbeTcp(host, port, timeoutMs = 1500) {
1572
+ return new Promise((resolvePromise) => {
1573
+ const socket = netConnect({ host, port });
1574
+ const done = (result) => {
1575
+ socket.destroy();
1576
+ resolvePromise(result);
1577
+ };
1578
+ socket.setTimeout(timeoutMs, () => done(false));
1579
+ socket.on("connect", () => done(true));
1580
+ socket.on("error", () => done(false));
1581
+ });
1582
+ }
1583
+
1016
1584
  /** Exit code of a spawned command; null when it could not launch (not on PATH) — the up.mjs pattern. */
1017
1585
  function run(ctx, cmd, args) {
1018
1586
  return new Promise((resolvePromise) => {
@@ -1028,6 +1596,28 @@ function run(ctx, cmd, args) {
1028
1596
  });
1029
1597
  }
1030
1598
 
1599
+ /**
1600
+ * Like runCapture() with the two streams SEPARATE, for a query whose answer is stdout alone (round 2, E2): podman
1601
+ * prints warnings on stderr on ordinary accounts, and merged into the answer they made a label read as someone else's.
1602
+ */
1603
+ function runQuery(ctx, cmd, args) {
1604
+ return new Promise((resolvePromise) => {
1605
+ let child;
1606
+ try {
1607
+ child = ctx.spawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
1608
+ } catch {
1609
+ resolvePromise({ code: null, stdout: "", stderr: "" });
1610
+ return;
1611
+ }
1612
+ let stdout = "";
1613
+ let stderr = "";
1614
+ child.stdout?.on("data", (d) => (stdout += d));
1615
+ child.stderr?.on("data", (d) => (stderr += d));
1616
+ child.on("error", () => resolvePromise({ code: null, stdout, stderr }));
1617
+ child.on("close", (code) => resolvePromise({ code, stdout, stderr }));
1618
+ });
1619
+ }
1620
+
1031
1621
  /** Like run() but with stdout+stderr captured, for read-only lookups (nssm status, is-active …). */
1032
1622
  function runCapture(ctx, cmd, args) {
1033
1623
  return new Promise((resolvePromise) => {