@edgehero/pi-dispatch 1.10.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +303 -150
- package/README.md +52 -0
- package/deploy/com.pi-dispatch.worker.plist +10 -4
- package/deploy/docker-compose.yml +49 -16
- package/deploy/egress-proxy.conf +32 -2
- package/deploy/nssm-install.cmd +12 -6
- package/deploy/pi-dispatch-egress-out.network +10 -0
- package/deploy/pi-dispatch-egress-proxy.container +50 -0
- package/deploy/pi-dispatch-netns-keeper.container +80 -0
- package/deploy/pi-dispatch-netns-keeper.network +18 -0
- package/deploy/pi-dispatch-valkey.container +51 -0
- package/deploy/pi-dispatch-valkey.network +16 -0
- package/deploy/receiver.service +6 -0
- package/deploy/worker-env-wrapper.cmd +12 -1
- package/deploy/worker-env-wrapper.sh +63 -37
- package/deploy/worker.service +18 -8
- package/package.json +15 -5
- package/src/azure-host.mjs +19 -0
- package/src/azure-identity.mjs +18 -2
- package/src/backend-conformance.mjs +71 -18
- package/src/backend-local.mjs +637 -21
- package/src/backend-podman.mjs +1168 -0
- package/src/backend-registry.mjs +86 -3
- package/src/backends.mjs +489 -37
- package/src/branch.mjs +7 -2
- package/src/cancel-cli.mjs +174 -0
- package/src/cancel-state.mjs +125 -0
- package/src/cli.mjs +188 -90
- package/src/config.mjs +503 -43
- package/src/connection.mjs +374 -8
- package/src/container-spec.mjs +102 -7
- package/src/daemon-facts.mjs +167 -0
- package/src/deployment-venue.mjs +158 -0
- package/src/docker-run.mjs +146 -15
- package/src/doctor.mjs +4756 -394
- package/src/egress-conf-copy.mjs +166 -0
- package/src/egress-proxy-state.mjs +151 -0
- package/src/egress.mjs +456 -25
- package/src/entry.mjs +27 -0
- package/src/env-allowlist.mjs +245 -40
- package/src/env-file.mjs +1869 -33
- package/src/exit-code.mjs +15 -0
- package/src/flow-gate.mjs +5 -3
- package/src/forgejo-host.mjs +19 -0
- package/src/forgejo-identity.mjs +21 -2
- package/src/get-token.mjs +67 -18
- package/src/git-dirty.mjs +9 -1
- package/src/git-hardening.mjs +33 -0
- package/src/github-app-setup.mjs +29 -12
- package/src/github-prompt.mjs +4 -1
- package/src/gitlab-host.mjs +19 -0
- package/src/gitlab-identity.mjs +19 -2
- package/src/host-pi.mjs +19 -3
- package/src/host-registry.mjs +29 -2
- package/src/identity.mjs +29 -4
- package/src/image-preflight.mjs +46 -11
- package/src/image-ref.mjs +21 -0
- package/src/index.mjs +363 -13
- package/src/init.mjs +197 -38
- package/src/job-user.mjs +252 -0
- package/src/json-duplicates.mjs +204 -0
- package/src/live-probes.mjs +1020 -0
- package/src/materialize.mjs +4 -11
- package/src/netns-keeper.mjs +264 -0
- package/src/on-failure.mjs +119 -0
- package/src/outbox.mjs +7 -0
- package/src/packages.mjs +2 -2
- package/src/podman-stack.mjs +1304 -0
- package/src/prepare-github.mjs +6 -6
- package/src/prepare-local.mjs +51 -17
- package/src/prepare.mjs +27 -6
- package/src/pricing.mjs +9 -5
- package/src/processor.mjs +506 -26
- package/src/provider-key.mjs +66 -0
- package/src/provider-steering.mjs +185 -0
- package/src/queue.mjs +35 -8
- package/src/redact.mjs +84 -0
- package/src/reserved-env.mjs +7 -3
- package/src/retention-sweep.mjs +178 -0
- package/src/run-container.mjs +181 -14
- package/src/run-history.mjs +105 -16
- package/src/runtime-observations.mjs +1152 -0
- package/src/runtime-settings.mjs +13 -8
- package/src/sandbox-cli.mjs +100 -95
- package/src/sandbox-store.mjs +612 -45
- package/src/sandbox.mjs +1459 -37
- package/src/schedules.mjs +16 -3
- package/src/secret-profiles.mjs +2 -1
- package/src/secrets.mjs +24 -6
- package/src/service-env.mjs +247 -0
- package/src/service.mjs +618 -28
- package/src/session-store.mjs +678 -53
- package/src/start.mjs +1348 -326
- package/src/subscriptions.mjs +7 -3
- package/src/transient.mjs +240 -0
- package/src/triggers-file.mjs +71 -15
- package/src/triggers.mjs +179 -19
- package/src/up.mjs +1399 -85
- package/src/valkey-auth.mjs +529 -0
- package/src/valkey-endpoint.mjs +367 -0
- package/src/watch-closer.mjs +158 -0
package/src/service.mjs
CHANGED
|
@@ -41,11 +41,19 @@
|
|
|
41
41
|
* the rest of the config is broken.
|
|
42
42
|
*/
|
|
43
43
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
44
|
-
import { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync } from "node:fs";
|
|
45
|
-
import {
|
|
44
|
+
import { chmodSync, existsSync, mkdirSync, readFileSync, realpathSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs";
|
|
45
|
+
import { lookup as dnsLookup } from "node:dns/promises";
|
|
46
|
+
import { connect as netConnect } from "node:net";
|
|
47
|
+
import { homedir, networkInterfaces, tmpdir, userInfo } from "node:os";
|
|
46
48
|
import { dirname, join, resolve } from "node:path";
|
|
47
49
|
import { fileURLToPath } from "node:url";
|
|
48
50
|
import { parseArgs } from "node:util";
|
|
51
|
+
import { parseBackendList, venuesOf } from "./backends.mjs";
|
|
52
|
+
import { sharedShellIgnored } from "./deployment-venue.mjs";
|
|
53
|
+
import { egressArmed, egressProxyName } from "./egress.mjs";
|
|
54
|
+
import { updateEnvFile } from "./env-file.mjs";
|
|
55
|
+
import { VALKEY_HEALTH_SCRIPT, VALKEY_PASSWORD_KEY, VALKEY_START_SCRIPT, dollarsDoubled, newValkeyPassword, valkeyPasswordDecision } from "./valkey-auth.mjs";
|
|
56
|
+
import { ALL_QUADLET_FILES, ALLOWLIST_PLACEHOLDER, NETNS_KEEPER, NETNS_KEEPER_FORMAT, judgeNetnsKeeper, keeperUnderRunningProxyHint, managerEnvRefusal, PROXY_CONF_PLACEHOLDER, QUADLET_FILES, applyStack, decideValkey, describeAction, passwdNameFrom, readSubuidRanges, readValkeyKeys, valkeySharedOn, VALKEY_SHARED_KEY, describeRollBack, journalWrite, rollBackWrites, foreignContainerRefusal, foreignContainers, lingerNote, planStack, proxyConfCopyPath, proxyRestartWarning, quadletDir, readLinger, readStackKeys, stackComponents, unknownContainerRefusal, userBusRefusal, valkeyEnvPath, valkeyPasswordRestartWarning, workerUnitDeps } from "./podman-stack.mjs";
|
|
49
57
|
|
|
50
58
|
// src/ is where this module lives in BOTH layouts (worker/src in a checkout,
|
|
51
59
|
// node_modules/@edgehero/pi-dispatch/src under npm). Deploy templates resolve one level up from it
|
|
@@ -74,6 +82,10 @@ function resolveReceiverStart() {
|
|
|
74
82
|
* worker/test/service.test.mjs asserts each one is still present in the real deploy/ file, so an edit
|
|
75
83
|
* to a template that would break the render fails the build instead of shipping a broken `service`.
|
|
76
84
|
*/
|
|
85
|
+
// The two worker-template literals a podman render rewrites (issue #430), spelled once for the pin table and the render.
|
|
86
|
+
const WORKER_DESCRIPTION = "Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)";
|
|
87
|
+
const WORKER_VALKEY_COMMENT = "# Valkey must be reachable (on this host, `pi-dispatch up` in the deployment folder starts it), but it is a separate\n# unit/container -- not ordered here since it may be remote.\n";
|
|
88
|
+
|
|
77
89
|
export const TEMPLATE_PINS = {
|
|
78
90
|
"worker.service": [
|
|
79
91
|
"ExecStart=/usr/bin/node worker/src/cli.mjs worker", // the WHOLE line → `<execPath> <cliPath> worker` (cli.mjs sits beside this module in src/ in both layouts)
|
|
@@ -81,6 +93,9 @@ export const TEMPLATE_PINS = {
|
|
|
81
93
|
"EnvironmentFile=/opt/pi-dispatch/.env", // → <deployDir>/.env — the operator's .env lives beside the units, never inside the package
|
|
82
94
|
"\nUser=pi\n", // the DIRECTIVE line (the header comment also says User=pi mid-line, hence the \n anchors): stripped for --user scope; rewritten to the invoking user for --system
|
|
83
95
|
"WantedBy=multi-user.target", // → default.target in user scope (multi-user.target never runs there)
|
|
96
|
+
"\nWants=network-online.target\n", // the anchor the podman venue's Wants=/After= on its Quadlet units go after (issue #430), user scope only
|
|
97
|
+
"Description=pi-dispatch worker (drains the job queue on the host; launches job containers via docker)", // → names rootless podman on a podman deployment (issue #430)
|
|
98
|
+
"# Valkey must be reachable (on this host, `pi-dispatch up` in the deployment folder starts it), but it is a separate\n# unit/container -- not ordered here since it may be remote.\n", // → says it IS ordered, when the Quadlet Valkey is installed
|
|
84
99
|
// Byte-for-byte survivors — semantics the render must not lose:
|
|
85
100
|
"RestartPreventExitStatus=2", // EXIT_POLICY is never restarted (a retry loop is a bill)
|
|
86
101
|
"StartLimitIntervalSec=60",
|
|
@@ -132,6 +147,52 @@ export const TEMPLATE_PINS = {
|
|
|
132
147
|
// a unit that starts fine and ignores the operator's secrets manager.
|
|
133
148
|
"worker-env-wrapper.sh": ["PI_ENV_SETUP"],
|
|
134
149
|
"worker-env-wrapper.cmd": ["PI_ENV_SETUP"],
|
|
150
|
+
// The podman venue's stack as Quadlet units (issue #430, podman-stack.mjs). Three are copied verbatim, so what is
|
|
151
|
+
// pinned is what the worker and the installer rely on by NAME: the container and network names the worker attaches
|
|
152
|
+
// by, the unit names the worker unit Wants=, the loopback-only port, and an explicit Network= in every .container
|
|
153
|
+
// (a containers.conf `netns = "host"` would otherwise put it in the host namespace, measured).
|
|
154
|
+
"pi-dispatch-valkey.network": ["NetworkName=pi-dispatch-valkey"],
|
|
155
|
+
"pi-dispatch-valkey.container": [
|
|
156
|
+
"ContainerName=pi-dispatch-valkey",
|
|
157
|
+
"Network=pi-dispatch-valkey.network",
|
|
158
|
+
"PublishPort=127.0.0.1:6379:6379",
|
|
159
|
+
"Volume=pi-dispatch-valkey-data:/data",
|
|
160
|
+
// Issue #468: the password from a 0600 file into the container's environment, then to valkey-server as
|
|
161
|
+
// configuration on stdin, never on an argv (tini keeps its argv as PID 1, readable by every account). The two lines
|
|
162
|
+
// are pinned to the ONE copy of each script in valkey-auth.mjs, `$` written `$$` for systemd.
|
|
163
|
+
"EnvironmentFile=%h/.config/pi-dispatch/valkey.env",
|
|
164
|
+
`Exec=sh -c '${dollarsDoubled(VALKEY_START_SCRIPT)}'`,
|
|
165
|
+
`HealthCmd=${dollarsDoubled(VALKEY_HEALTH_SCRIPT)}`,
|
|
166
|
+
"WantedBy=default.target",
|
|
167
|
+
],
|
|
168
|
+
"pi-dispatch-egress-out.network": ["NetworkName=pi-dispatch-egress-out"],
|
|
169
|
+
"pi-dispatch-egress-proxy.container": [
|
|
170
|
+
`Volume=${PROXY_CONF_PLACEHOLDER}:/etc/squid/squid.conf:ro,z`, // → the account-owned COPY of the package's egress-proxy.conf (~/.config/pi-dispatch/egress-proxy.conf, podman-stack.mjs proxyConfCopyPath), because `z` cannot relabel a root-owned package file (measured)
|
|
171
|
+
`Volume=${ALLOWLIST_PLACEHOLDER}:/etc/pi-dispatch/allowlist.conf:ro,z`, // → <deployDir>/egress-allowlist.conf, the list `init` scaffolds
|
|
172
|
+
"ContainerName=pi-dispatch-egress-proxy",
|
|
173
|
+
"Network=pi-dispatch-egress-out.network",
|
|
174
|
+
"WantedBy=default.target",
|
|
175
|
+
],
|
|
176
|
+
// The rootless network keeper (issue #458), copied verbatim. Pinned: the names (outside every sweep's prefix, and
|
|
177
|
+
// what doctor reads), its own network with no route and no DNS, and every line that keeps it from widening anything
|
|
178
|
+
// (podman-stack.test.mjs reads the same lines back as the argv they generate).
|
|
179
|
+
"pi-dispatch-netns-keeper.network": ["NetworkName=pi-dispatch-netns-keeper", "Internal=true", "DisableDNS=true"],
|
|
180
|
+
"pi-dispatch-netns-keeper.container": [
|
|
181
|
+
"ContainerName=pi-dispatch-netns-keeper",
|
|
182
|
+
"Network=pi-dispatch-netns-keeper.network",
|
|
183
|
+
"ReadOnly=true",
|
|
184
|
+
"DropCapability=all",
|
|
185
|
+
"NoNewPrivileges=true",
|
|
186
|
+
"User=65534",
|
|
187
|
+
"Group=65534",
|
|
188
|
+
"RunInit=true",
|
|
189
|
+
"ExecStartPre=/usr/bin/podman network create --ignore --disable-dns --internal pi-dispatch-netns-keeper",
|
|
190
|
+
"Restart=always",
|
|
191
|
+
"RestartSec=1s",
|
|
192
|
+
"StartLimitIntervalSec=0",
|
|
193
|
+
"SuccessExitStatus=143",
|
|
194
|
+
"WantedBy=default.target",
|
|
195
|
+
],
|
|
135
196
|
};
|
|
136
197
|
|
|
137
198
|
/**
|
|
@@ -231,6 +292,19 @@ export function readUnitSeam(text, platform) {
|
|
|
231
292
|
return { setup: clip(grab(readers.setup)), deployDir: clip(grab(readers.deployDir)) };
|
|
232
293
|
}
|
|
233
294
|
|
|
295
|
+
/**
|
|
296
|
+
* The account a SYSTEM unit runs the worker as (`User=`), or `null` when it names none (issue #341). Only systemd has
|
|
297
|
+
* one: a user-scope unit, a launchd agent and an nssm service run as whoever installed them. Separate from
|
|
298
|
+
* `readUnitSeam` because it is not part of the render round trip (a `--system` render writes the invoking user, a
|
|
299
|
+
* `--user` render strips the line). Read as systemd does: whitespace around `=` allowed, and the LAST assignment wins.
|
|
300
|
+
*/
|
|
301
|
+
export function readUnitUser(text, platform) {
|
|
302
|
+
if (platform !== "linux" || typeof text !== "string") return null;
|
|
303
|
+
const hits = [...text.replace(/\0/g, "").matchAll(/^[ \t]*User[ \t]*=[ \t]*(.*?)[ \t]*\r?$/gm)];
|
|
304
|
+
const value = hits.at(-1)?.[1];
|
|
305
|
+
return value ? value : null;
|
|
306
|
+
}
|
|
307
|
+
|
|
234
308
|
const SUBCOMMANDS = new Set(["render", "install", "uninstall", "status", "start", "stop", "restart"]);
|
|
235
309
|
|
|
236
310
|
const SERVICE_USAGE = `pi-dispatch service — run the worker (or --receiver) as an OS service, rendered for THIS host
|
|
@@ -269,13 +343,26 @@ export async function runService(argv = [], deps = {}) {
|
|
|
269
343
|
// this command. An operator changes it by running the command as someone else, not by declaring it.
|
|
270
344
|
user = env.USER || userInfo().username,
|
|
271
345
|
tmp = tmpdir(),
|
|
272
|
-
fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync },
|
|
346
|
+
fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync, chmodSync, statSync, renameSync, realpathSync },
|
|
273
347
|
spawn = nodeSpawn,
|
|
274
348
|
out = (s) => process.stdout.write(s),
|
|
275
349
|
err = (s) => process.stderr.write(s),
|
|
276
350
|
sleep = (ms) => new Promise((r) => setTimeout(r, ms)),
|
|
277
351
|
now = () => Date.now(),
|
|
278
352
|
queue = null, // test seam; production builds one lazily in doRestart from VALKEY_URL
|
|
353
|
+
// Is anything on 127.0.0.1:6379? Asked only on a podman deployment, to decide whether the Quadlet Valkey is
|
|
354
|
+
// wanted (issue #430); the same plain TCP connect `up` uses, for up's reason. Issue #464: asked of every address
|
|
355
|
+
// VALKEY_URL's host resolves to (`lookup`, as the worker's client resolves it), and `interfaces` says which of
|
|
356
|
+
// them are this host's.
|
|
357
|
+
probeTcp = defaultProbeTcp,
|
|
358
|
+
lookup = (host, opts) => dnsLookup(host, opts),
|
|
359
|
+
interfaces = networkInterfaces,
|
|
360
|
+
// Issue #464 (gate round 2): how `getsubids` is run for this account's subordinate ranges (null: not installed).
|
|
361
|
+
runSync = undefined,
|
|
362
|
+
// PR #463: how a path resolves through symlinks, for the manager-environment rule (a symlinked home).
|
|
363
|
+
realpath = (p) => realpathSync(p),
|
|
364
|
+
// Issue #468: how a new Valkey password is made (a test's seam; 32 random bytes, hex, never shown).
|
|
365
|
+
newPassword = newValkeyPassword,
|
|
279
366
|
} = deps;
|
|
280
367
|
|
|
281
368
|
let values, positionals;
|
|
@@ -338,6 +425,12 @@ export async function runService(argv = [], deps = {}) {
|
|
|
338
425
|
sleep,
|
|
339
426
|
now,
|
|
340
427
|
queue,
|
|
428
|
+
probeTcp,
|
|
429
|
+
lookup,
|
|
430
|
+
interfaces,
|
|
431
|
+
runSync,
|
|
432
|
+
realpath,
|
|
433
|
+
newPassword,
|
|
341
434
|
which: values.receiver ? "receiver" : "worker",
|
|
342
435
|
scope: platform === "linux" && values.system ? "system" : "user",
|
|
343
436
|
force: values.force,
|
|
@@ -363,7 +456,7 @@ export async function runService(argv = [], deps = {}) {
|
|
|
363
456
|
return doRender(ctx);
|
|
364
457
|
case "install":
|
|
365
458
|
// --print is implied for render and opt-in here: see what will be written, then write it.
|
|
366
|
-
if (values.print) doRender(ctx);
|
|
459
|
+
if (values.print) await doRender(ctx);
|
|
367
460
|
return doInstall(ctx);
|
|
368
461
|
case "uninstall":
|
|
369
462
|
return doUninstall(ctx);
|
|
@@ -511,7 +604,7 @@ function composeEnvSetupExec(setup, argv) {
|
|
|
511
604
|
* known literals (see TEMPLATE_PINS); everything else — RestartPreventExitStatus=2, the StartLimit
|
|
512
605
|
* crash-loop bound, KillSignal, TimeoutStopSec — passes through byte-for-byte.
|
|
513
606
|
*/
|
|
514
|
-
function renderLinuxUnit(ctx) {
|
|
607
|
+
function renderLinuxUnit(ctx, stackUnits = [], venues = null) {
|
|
515
608
|
const template = ctx.which === "receiver" ? "receiver.service" : "worker.service";
|
|
516
609
|
// The ExecStart line is replaced WHOLE, not path-by-path: the template's script path is relative
|
|
517
610
|
// to a repo-root WorkingDirectory that only a checkout has. The rendered unit points at absolute
|
|
@@ -549,6 +642,19 @@ function renderLinuxUnit(ctx) {
|
|
|
549
642
|
// symlink into a .wants/ directory no user-instance boot ever walks — enabled but never
|
|
550
643
|
// started. default.target is the user manager's boot target.
|
|
551
644
|
unit = unit.replace("WantedBy=multi-user.target", "WantedBy=default.target");
|
|
645
|
+
// The podman venue's Quadlet units (issue #430), which live in this same user manager, so the worker can be
|
|
646
|
+
// ordered after them. Nothing is added when there are none, which keeps every other render byte-identical.
|
|
647
|
+
const deps = workerUnitDeps(stackUnits);
|
|
648
|
+
if (deps) unit = unit.replace("\nWants=network-online.target\n", () => `\nWants=network-online.target\n${deps}`);
|
|
649
|
+
// The template's own words, made true for a podman deployment (round 2 nit): its Description says jobs launch
|
|
650
|
+
// via docker, and its comment says Valkey is not ordered here. Both anchors are TEMPLATE_PINS.
|
|
651
|
+
if (venues?.podmanUsed) {
|
|
652
|
+
const runtime = venues.localUsed ? "docker and rootless podman" : "rootless podman";
|
|
653
|
+
unit = unit.replace(WORKER_DESCRIPTION, () => `Description=pi-dispatch worker (drains the job queue on the host; launches job containers via ${runtime})`);
|
|
654
|
+
if (stackUnits.includes(QUADLET_FILES.valkey.unit)) {
|
|
655
|
+
unit = unit.replace(WORKER_VALKEY_COMMENT, () => "# Valkey is the podman venue's Quadlet unit in this same user manager, so the worker is ordered after it\n# (the Wants=/After= lines `pi-dispatch service install` added below).\n");
|
|
656
|
+
}
|
|
657
|
+
}
|
|
552
658
|
} else {
|
|
553
659
|
unit = unit.replace(/^User=pi$/m, () => `User=${ctx.user}`);
|
|
554
660
|
}
|
|
@@ -558,8 +664,11 @@ function renderLinuxUnit(ctx) {
|
|
|
558
664
|
/**
|
|
559
665
|
* Render the launchd plist for this host. For --receiver the worker plist is DERIVED, not a second
|
|
560
666
|
* template: same KeepAlive/ExitTimeOut shape, label and log names swapped, and the shared wrapper given
|
|
561
|
-
* the receiver's exec argv instead of the worker's. The wrapper's exit-2 conversion is
|
|
562
|
-
* receiver
|
|
667
|
+
* the receiver's exec argv instead of the worker's. The wrapper's exit-2 conversion is LOAD-BEARING for
|
|
668
|
+
* the receiver too, not the no-op an earlier version of this comment claimed: since
|
|
669
|
+
* receiver/src/cli.mjs gained entryExitCode, a determinate receiver refusal exits EXIT_POLICY (2), and
|
|
670
|
+
* the wrapper's conversion to a clean 0 is exactly what KeepAlive/SuccessfulExit=false reads as
|
|
671
|
+
* "leave it stopped" -- launchd's only way to spell RestartPreventExitStatus=2.
|
|
563
672
|
*/
|
|
564
673
|
function renderPlist(ctx) {
|
|
565
674
|
// Substituted before anything composed goes in (subDeployDir): the two anchors below are therefore
|
|
@@ -650,7 +759,151 @@ function nssmSequence(ctx) {
|
|
|
650
759
|
};
|
|
651
760
|
}
|
|
652
761
|
|
|
653
|
-
|
|
762
|
+
/**
|
|
763
|
+
* Does this worker deployment run the podman venue, per the `.env` its unit loads? `{ used: false }`, `{ error }`,
|
|
764
|
+
* or `{ used: true, venues, env }`. Only the worker has a stack: the receiver runs no job and needs no Podman.
|
|
765
|
+
*
|
|
766
|
+
* The keys come from the deployment's `.env`, the file the unit's `EnvironmentFile=` loads, read the way that loader
|
|
767
|
+
* reads it (`readStackKeys`, shared with `up`). NOT this shell's environment: nothing in this project loads `.env` into
|
|
768
|
+
* a process (docs/secrets.md), so a shell that exports PI_BACKENDS=podman says nothing about what the SERVICE will
|
|
769
|
+
* run, and installing Quadlet units off it would stand up a stack for a worker that then runs docker. A key an
|
|
770
|
+
* `--env-setup` script exports is invisible here for the same reason, which docs/podman.md says.
|
|
771
|
+
*
|
|
772
|
+
* On macOS and Windows the answer decides only a NOTE (the venue runs only on Linux, so there is no stack to
|
|
773
|
+
* install), so it is read with that platform's own loader (the wrapper sources the file with sh; the cmd wrapper
|
|
774
|
+
* splits it) and a file it cannot read there is not a refusal: refusing an install over a key that could change
|
|
775
|
+
* nothing about it would be a refusal with no remedy.
|
|
776
|
+
*/
|
|
777
|
+
function podmanVenue(ctx) {
|
|
778
|
+
if (ctx.which !== "worker") return { used: false };
|
|
779
|
+
const envPath = join(ctx.deployDir, ".env");
|
|
780
|
+
const linux = ctx.platform === "linux";
|
|
781
|
+
let fileKeys = {};
|
|
782
|
+
if (ctx.fs.existsSync(envPath)) {
|
|
783
|
+
let text;
|
|
784
|
+
try {
|
|
785
|
+
// Bytes, not text: `readStackKeys` checks what systemd refuses to load before it decodes (issue #447).
|
|
786
|
+
text = ctx.fs.readFileSync(envPath);
|
|
787
|
+
} catch (err) {
|
|
788
|
+
if (!linux) return { used: false };
|
|
789
|
+
return { error: `cannot read ${envPath} to learn whether this deployment runs the podman venue: ${err?.message}` };
|
|
790
|
+
}
|
|
791
|
+
const loader = linux ? "systemd" : ctx.platform === "darwin" ? "shell" : "cmd";
|
|
792
|
+
const read = readStackKeys(text, { loader, path: envPath });
|
|
793
|
+
if (read.error) return linux ? { error: read.error } : { used: false };
|
|
794
|
+
fileKeys = read.keys;
|
|
795
|
+
// Issue #464: where the worker's queue is, and whether a Valkey another uid holds may be it (PI_VALKEY_SHARED), as
|
|
796
|
+
// the service reads them, for the Valkey rule (`decideValkey`). A line the loaders read differently is refused,
|
|
797
|
+
// naming what in it is the problem, rather than guessed at.
|
|
798
|
+
const valkey = readValkeyKeys(text, { loader, path: envPath });
|
|
799
|
+
if (valkey.error) return linux ? { error: valkey.error } : { used: false };
|
|
800
|
+
fileKeys = { ...fileKeys, ...valkey.keys };
|
|
801
|
+
}
|
|
802
|
+
// Refused here, unlike doctor's venuesOf, which reads an unparseable list as `local`: doctor then reports the parse,
|
|
803
|
+
// but an install that guessed would write a unit and a stack for a deployment the worker refuses to boot.
|
|
804
|
+
try {
|
|
805
|
+
parseBackendList(fileKeys.PI_BACKENDS);
|
|
806
|
+
} catch (err) {
|
|
807
|
+
return linux ? { error: `${envPath}: ${err.message}` } : { used: false };
|
|
808
|
+
}
|
|
809
|
+
const venues = venuesOf(fileKeys);
|
|
810
|
+
if (!venues.podmanUsed) return { used: false };
|
|
811
|
+
return { used: true, venues, env: fileKeys };
|
|
812
|
+
}
|
|
813
|
+
|
|
814
|
+
/**
|
|
815
|
+
* The refusal for the podman venue in SYSTEM scope. Not "podman cannot run under a system unit": a system unit with
|
|
816
|
+
* `User=` runs rootless Podman fine (measured, docs/podman.md step 2). The stack is the problem. Its Quadlet units
|
|
817
|
+
* belong to the account's own user manager, and a system unit cannot `Wants=`/`After=` a unit of another manager, so
|
|
818
|
+
* the worker would race its own queue at every boot; and installing them means writing into a user's config from a
|
|
819
|
+
* command whose system-scope doctrine is to write nothing. User scope with linger does both properly.
|
|
820
|
+
*/
|
|
821
|
+
function refusePodmanSystemScope(ctx) {
|
|
822
|
+
return fail(
|
|
823
|
+
ctx.err,
|
|
824
|
+
`PI_BACKENDS in ${join(ctx.deployDir, ".env")} lists podman, and the podman venue's stack (Valkey, the egress proxy) runs as Quadlet units in the worker account's OWN user manager, which a system unit cannot order itself after. Install in user scope (drop --system) with linger on: sudo loginctl enable-linger ${ctx.user}\nOr keep a hand-written system unit and start the stack by hand (docs/podman.md, steps 6 and 7).`,
|
|
825
|
+
);
|
|
826
|
+
}
|
|
827
|
+
|
|
828
|
+
/**
|
|
829
|
+
* The podman venue's stack for this install: which parts, rendered, with the actions that install them. `{ error }`
|
|
830
|
+
* for what cannot be installed, else `{ components, plan, notes }`. The Valkey rule differs from `up`'s on purpose:
|
|
831
|
+
* `up` starts what is not running, while an install also keeps (and orders the worker after) a Valkey unit an
|
|
832
|
+
* earlier run already installed, even though that Valkey is now the thing listening.
|
|
833
|
+
*/
|
|
834
|
+
async function podmanStackFor(ctx, venue, { readKeeper = false } = {}) {
|
|
835
|
+
let armed;
|
|
836
|
+
try {
|
|
837
|
+
armed = egressArmed(venue.env);
|
|
838
|
+
} catch (err) {
|
|
839
|
+
return { error: `${join(ctx.deployDir, ".env")}: ${err.message}` };
|
|
840
|
+
}
|
|
841
|
+
const installed = ctx.fs.existsSync(join(quadletDir(ctx.home), QUADLET_FILES.valkey.file));
|
|
842
|
+
// Issue #464: a listener on any address VALKEY_URL reaches is taken to be this deployment's Valkey only when it is
|
|
843
|
+
// this account's (its uid, or a subordinate uid its containers run as), or PI_VALKEY_SHARED=1 in .env says it is
|
|
844
|
+
// shared on purpose; anything else is refused, not forceably (`stackRefusal`).
|
|
845
|
+
const valkey = await decideValkey({
|
|
846
|
+
venues: venue.venues,
|
|
847
|
+
url: venue.env.VALKEY_URL,
|
|
848
|
+
installed,
|
|
849
|
+
probeTcp: ctx.probeTcp,
|
|
850
|
+
lookup: ctx.lookup,
|
|
851
|
+
interfaces: ctx.interfaces,
|
|
852
|
+
euid: ctx.euid,
|
|
853
|
+
user: ctx.user,
|
|
854
|
+
shared: valkeySharedOn(venue.env[VALKEY_SHARED_KEY]),
|
|
855
|
+
subuids: readSubuidRanges({ user: ctx.user, euid: ctx.euid, fs: ctx.fs, run: ctx.runSync }),
|
|
856
|
+
fs: ctx.fs,
|
|
857
|
+
ownerName: (uid) => passwdNameFrom(ctx.fs, uid),
|
|
858
|
+
envPath: join(ctx.deployDir, ".env"),
|
|
859
|
+
});
|
|
860
|
+
if (valkey.error) return { error: valkey.error };
|
|
861
|
+
// Issue #468: the password the Quadlet Valkey starts with, when this install includes it: the deployment's own
|
|
862
|
+
// VALKEY_PASSWORD, or a new one this install writes into .env (never over a value), unless the Valkey is shared or
|
|
863
|
+
// the operator's own (`valkeyPasswordDecision`). Generated here, written by doInstall once nothing refuses; `render`
|
|
864
|
+
// writes nothing and shows the password file by its name alone.
|
|
865
|
+
let valkeyPassword = null;
|
|
866
|
+
let pendingPassword = null;
|
|
867
|
+
const passwordNotes = [];
|
|
868
|
+
if (valkey.include) {
|
|
869
|
+
const decided = valkeyPasswordDecision(venue.env, { envPath: join(ctx.deployDir, ".env") });
|
|
870
|
+
if (decided.error) return { error: decided.error };
|
|
871
|
+
if (decided.note) passwordNotes.push(decided.note);
|
|
872
|
+
if (decided.generate) pendingPassword = ctx.newPassword();
|
|
873
|
+
valkeyPassword = decided.password ?? pendingPassword;
|
|
874
|
+
}
|
|
875
|
+
const components = stackComponents({ venues: venue.venues, env: venue.env, includeValkey: valkey.include, armed, valkeyPort: valkey.port, valkeyPassword });
|
|
876
|
+
components.notes.push(...valkey.notes, ...passwordNotes);
|
|
877
|
+
// Gate round 2: the service reads PI_VALKEY_SHARED from .env alone, so a shell export says nothing; named, not honoured.
|
|
878
|
+
if (typeof ctx.env[VALKEY_SHARED_KEY] === "string") components.notes.push(sharedShellIgnored(ctx.env[VALKEY_SHARED_KEY], join(ctx.deployDir, ".env")));
|
|
879
|
+
// The keeper, read as `up` reads it (PR #463 final review), install only (render spawns nothing): a Quadlet keeper
|
|
880
|
+
// whose file is already there and unchanged but that does not hold (stopped, paused, off its bridge) is RESTARTED,
|
|
881
|
+
// since `start` over it is a no-op for a paused one and would say nothing of the proxy either way; the plan then
|
|
882
|
+
// restarts our proxy after it (planStack), or the install names the operator's own proxy to restart.
|
|
883
|
+
let restartUnits = [];
|
|
884
|
+
if (readKeeper && components.keeper && ctx.fs.existsSync(join(quadletDir(ctx.home), QUADLET_FILES.keeper.file))) {
|
|
885
|
+
const keeper = judgeNetnsKeeper(await runQuery(ctx, "podman", ["inspect", NETNS_KEEPER_FORMAT, NETNS_KEEPER]));
|
|
886
|
+
if (!keeper.holds) restartUnits = [QUADLET_FILES.keeper.unit];
|
|
887
|
+
}
|
|
888
|
+
let plan = planStack({ components, templatesDir: ctx.templatesDir, deployDir: ctx.deployDir, home: ctx.home, fs: ctx.fs, restartUnits });
|
|
889
|
+
if (plan.error) return { error: plan.error };
|
|
890
|
+
// Issue #468: a Valkey that already ran and now gets a new password (an upgrade from a deployment that had none) is one
|
|
891
|
+
// its running clients can no longer talk to, so the worker and the receiver units this account runs are restarted
|
|
892
|
+
// after it. Asked only then (install only; render spawns nothing), and only of units the manager says are active.
|
|
893
|
+
const secret = plan.files.find((f) => f.kind === "secret");
|
|
894
|
+
const valkeyFile = plan.files.find((f) => f.unit === QUADLET_FILES.valkey.unit && f.path.endsWith(".container"));
|
|
895
|
+
if (readKeeper && secret && secret.state !== "same" && valkeyFile && valkeyFile.state !== "new") {
|
|
896
|
+
const active = [];
|
|
897
|
+
for (const unit of ["pi-dispatch-worker.service", "pi-dispatch-receiver.service"]) {
|
|
898
|
+
const res = await runQuery(ctx, "systemctl", ["--user", "is-active", unit]);
|
|
899
|
+
if (res.code === 0 && String(res.stdout ?? "").trim() === "active") active.push(unit);
|
|
900
|
+
}
|
|
901
|
+
if (active.length > 0) plan = planStack({ components, templatesDir: ctx.templatesDir, deployDir: ctx.deployDir, home: ctx.home, fs: ctx.fs, restartUnits, restartAfterValkey: active });
|
|
902
|
+
}
|
|
903
|
+
return { components, plan, notes: components.notes, venues: venue.venues, proxy: egressProxyName(venue.env), valkeyRefusal: valkey.refusal, pendingPassword };
|
|
904
|
+
}
|
|
905
|
+
|
|
906
|
+
async function doRender(ctx) {
|
|
654
907
|
if (ctx.which === "receiver" && !ctx.receiverStart) return refuseMissingReceiver(ctx);
|
|
655
908
|
const paths = unitPaths(ctx);
|
|
656
909
|
if (ctx.platform === "darwin") {
|
|
@@ -667,8 +920,26 @@ function doRender(ctx) {
|
|
|
667
920
|
return 0;
|
|
668
921
|
}
|
|
669
922
|
if (ctx.platform === "linux") {
|
|
923
|
+
const venue = podmanVenue(ctx);
|
|
924
|
+
if (venue.error) return fail(ctx.err, venue.error);
|
|
925
|
+
let stack = null;
|
|
926
|
+
if (venue.used) {
|
|
927
|
+
if (ctx.scope === "system") return refusePodmanSystemScope(ctx);
|
|
928
|
+
stack = await podmanStackFor(ctx, venue);
|
|
929
|
+
if (stack.error) return fail(ctx.err, stack.error);
|
|
930
|
+
}
|
|
670
931
|
ctx.out(`# → ${paths.installPath}\n`);
|
|
671
|
-
ctx.out(renderLinuxUnit(ctx));
|
|
932
|
+
ctx.out(renderLinuxUnit(ctx, stack?.plan.start ?? [], stack?.venues ?? null));
|
|
933
|
+
// The proxy's rules copy is named, not printed (round 3 nit): it is the package's egress-proxy.conf verbatim, and
|
|
934
|
+
// 4.7 KB of squid configuration under a "Quadlet" heading read as a unit file.
|
|
935
|
+
for (const f of stack?.plan.files ?? []) {
|
|
936
|
+
if (f.kind === "conf") ctx.out(`\n# → ${f.path} (the egress proxy's rules: install copies the package's egress-proxy.conf here, unchanged)\n`);
|
|
937
|
+
// Issue #468: the password file is NAMED, never printed: render output lands in scrollbacks and bug reports.
|
|
938
|
+
else if (f.kind === "secret") ctx.out(`\n# → ${f.path} (mode 0600: ${VALKEY_PASSWORD_KEY} for the Quadlet Valkey, ${f.text.includes(`${VALKEY_PASSWORD_KEY}=\n`) ? "empty, so it starts without a password" : "from .env or generated into it at install; the value is not shown"})\n`);
|
|
939
|
+
else ctx.out(`\n# → ${f.path} (Quadlet, podman venue)\n${f.text}`);
|
|
940
|
+
}
|
|
941
|
+
for (const note of stack?.notes ?? []) ctx.out(`# note: ${note}\n`);
|
|
942
|
+
if (stack?.valkeyRefusal) ctx.out(`# note: install refuses this: ${stack.valkeyRefusal.text}\n`);
|
|
672
943
|
return 0;
|
|
673
944
|
}
|
|
674
945
|
const { service, commands } = nssmSequence(ctx);
|
|
@@ -697,9 +968,27 @@ async function doInstall(ctx) {
|
|
|
697
968
|
}
|
|
698
969
|
}
|
|
699
970
|
|
|
700
|
-
|
|
701
|
-
if (
|
|
702
|
-
|
|
971
|
+
const venue = podmanVenue(ctx);
|
|
972
|
+
if (venue.error) return fail(ctx.err, venue.error);
|
|
973
|
+
if (ctx.platform === "darwin" || ctx.platform === "win32") {
|
|
974
|
+
const code = ctx.platform === "darwin" ? await installDarwin(ctx, paths) : await installWindows(ctx);
|
|
975
|
+
// Said, never silently skipped: the venue refuses every host that is not Linux (podman-platform), so there is
|
|
976
|
+
// no stack to install here, and a deployment listing it should know why nothing about Podman happened.
|
|
977
|
+
if (code === 0 && venue.used) ctx.out("note: PI_BACKENDS lists podman, and the podman venue runs only on Linux (it refuses this host as podman-platform), so no podman stack was installed here\n");
|
|
978
|
+
return code;
|
|
979
|
+
}
|
|
980
|
+
if (ctx.scope === "system") return venue.used ? refusePodmanSystemScope(ctx) : installLinuxSystem(ctx, paths);
|
|
981
|
+
let stack = null;
|
|
982
|
+
let foreign = [];
|
|
983
|
+
if (venue.used) {
|
|
984
|
+
stack = await podmanStackFor(ctx, venue, { readKeeper: true });
|
|
985
|
+
if (stack.error) return fail(ctx.err, stack.error);
|
|
986
|
+
// Every stack refusal first, the unit's own among them, so installLinuxUser's unit check never fires alone.
|
|
987
|
+
const refused = await stackRefusal(ctx, paths, stack);
|
|
988
|
+
if (refused.error) return fail(ctx.err, refused.error);
|
|
989
|
+
foreign = refused.foreign;
|
|
990
|
+
}
|
|
991
|
+
return installLinuxUser(ctx, paths, stack, foreign);
|
|
703
992
|
}
|
|
704
993
|
|
|
705
994
|
async function installDarwin(ctx, paths) {
|
|
@@ -721,11 +1010,18 @@ async function installDarwin(ctx, paths) {
|
|
|
721
1010
|
// the old copy out first. "Not loaded" is a fine answer — the nonzero exit is ignored.
|
|
722
1011
|
await run(ctx, "launchctl", ["bootout", `gui/${ctx.euid}/${paths.name}`]);
|
|
723
1012
|
}
|
|
724
|
-
|
|
725
|
-
//
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
1013
|
+
// Issue #464: a write that fails is said, with what this run left, rather than thrown out of the command. The plist
|
|
1014
|
+
// is the one file this path writes, so a failure leaves none of ours (a failed write is put back like the Linux path's).
|
|
1015
|
+
const journal = [];
|
|
1016
|
+
try {
|
|
1017
|
+
ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
|
|
1018
|
+
// launchd creates the StandardOutPath FILES but not their parent directory: without this the job
|
|
1019
|
+
// spawns and dies with its error unwritable. Done at install, not render: render stays read-only.
|
|
1020
|
+
ctx.fs.mkdirSync(join(ctx.deployDir, "logs"), { recursive: true });
|
|
1021
|
+
journalWrite(ctx.fs, journal, paths.installPath, renderPlist(ctx));
|
|
1022
|
+
} catch (err) {
|
|
1023
|
+
return fail(ctx.err, `could not write ${paths.installPath} (${err?.message ?? err}), so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}`);
|
|
1024
|
+
}
|
|
729
1025
|
const bootstrap = await run(ctx, "launchctl", ["bootstrap", `gui/${ctx.euid}`, paths.installPath]);
|
|
730
1026
|
if (bootstrap !== 0) {
|
|
731
1027
|
return fail(ctx.err, `launchctl bootstrap failed (exit ${bootstrap}) — the plist is written; retry by hand: launchctl bootstrap gui/${ctx.euid} ${paths.installPath}`);
|
|
@@ -740,25 +1036,176 @@ async function installDarwin(ctx, paths) {
|
|
|
740
1036
|
return 0;
|
|
741
1037
|
}
|
|
742
1038
|
|
|
743
|
-
|
|
1039
|
+
/**
|
|
1040
|
+
* Every reason the podman venue's stack refuses an install, gathered BEFORE anything is written (round 2, E9): the
|
|
1041
|
+
* worker unit existing used to be the only thing said, so an operator re-ran with --force and then met a replaced
|
|
1042
|
+
* container and a restarted proxy nobody had mentioned. `--force` is consent to exactly the list printed here. Three
|
|
1043
|
+
* reasons are not forceable: a missing allowlist, a container whose state could not be read, and no user manager.
|
|
1044
|
+
* Returns `{ error }` or `{ foreign }` (the foreign containers --force will replace, for the warnings).
|
|
1045
|
+
*/
|
|
1046
|
+
async function stackRefusal(ctx, paths, stack) {
|
|
1047
|
+
const blocking = [];
|
|
1048
|
+
const forceable = [];
|
|
1049
|
+
if (ctx.fs.existsSync(paths.installPath)) forceable.push(`${paths.installPath} already exists (same non-clobber contract as init); --force replaces it`);
|
|
1050
|
+
const changed = stack.plan.files.filter((f) => f.state === "changed");
|
|
1051
|
+
if (changed.length > 0) forceable.push(`${changed.map((f) => f.path).join(", ")} already ${changed.length === 1 ? "exists" : "exist"} with other content than this version renders; --force replaces ${changed.length === 1 ? "it" : "them"} and restarts ${(stack.plan.restart ?? []).join(" ") || "nothing"}${proxyRestartWarning(stack.plan) ? ` (${proxyRestartWarning(stack.plan)})` : ""}`);
|
|
1052
|
+
// Issue #464: a Valkey VALKEY_URL reaches that is not this account's. NOT forceable: --force is how an operator replaces
|
|
1053
|
+
// a changed file, and taking another account's queue must never ride along with that. The opt-in is its own named key.
|
|
1054
|
+
if (stack.valkeyRefusal) blocking.push(stack.valkeyRefusal.text);
|
|
1055
|
+
const allowlist = join(ctx.deployDir, "egress-allowlist.conf");
|
|
1056
|
+
if (stack.components.proxy && !ctx.fs.existsSync(allowlist)) {
|
|
1057
|
+
blocking.push(`the egress policy is on, and ${allowlist} does not exist: a proxy unit mounting a missing file makes Podman create a DIRECTORY there and squid fail confusingly. Run \`pi-dispatch init\` in ${ctx.deployDir} first (it never overwrites), or set PI_EGRESS=0 in .env to opt out of the policy`);
|
|
1058
|
+
}
|
|
1059
|
+
const bus = stack.plan.actions.length > 0 ? userBusRefusal({ env: ctx.env, user: ctx.user, euid: ctx.euid }) : null;
|
|
1060
|
+
if (bus) blocking.push(bus);
|
|
1061
|
+
// The manager's own XDG_RUNTIME_DIR and XDG_CONFIG_HOME (PR #463 round 3, measured): another account's makes every
|
|
1062
|
+
// unit this installs fail to start, or not exist at all. Asked only where the manager can be asked.
|
|
1063
|
+
if (!bus && stack.plan.actions.length > 0) {
|
|
1064
|
+
const managerEnv = managerEnvRefusal(await runQuery(ctx, "systemctl", ["--user", "show-environment"]), { home: ctx.home, euid: ctx.euid, user: ctx.user, realpath: ctx.realpath });
|
|
1065
|
+
if (managerEnv) blocking.push(managerEnv);
|
|
1066
|
+
}
|
|
1067
|
+
// A container of the unit's name that the unit does not own would be removed by its `podman run --replace`
|
|
1068
|
+
// (issue #430 review): refused unless --force says to replace it, and then said out loud.
|
|
1069
|
+
const containers = await foreignContainers(stack.plan, (cmd, args) => runQuery(ctx, cmd, args));
|
|
1070
|
+
const foreign = containers.found;
|
|
1071
|
+
if (containers.unknown.length > 0) blocking.push(unknownContainerRefusal(containers.unknown));
|
|
1072
|
+
if (foreign.length > 0) forceable.push(foreignContainerRefusal(foreign, { forceHint: "pass --force to let the Quadlet unit replace it" }));
|
|
1073
|
+
if (blocking.length > 0 || (forceable.length > 0 && !ctx.force)) {
|
|
1074
|
+
const lines = [...blocking, ...(ctx.force ? [] : forceable)];
|
|
1075
|
+
// Worded from the counts, since both lists can hold several (round 3 nit): the blocking reasons come first.
|
|
1076
|
+
const byHand = blocking.length === 1 ? "The first item must be fixed by hand" : `The first ${blocking.length} items must be fixed by hand`;
|
|
1077
|
+
const tail = blocking.length === 0 ? "\n--force accepts every item above at once." : forceable.length > 0 && !ctx.force ? `\n${byHand}; --force accepts the rest.` : `\n${blocking.length === 1 ? "It" : "Each"} must be fixed by hand; --force does not apply.`;
|
|
1078
|
+
return { error: `${lines.length === 1 ? lines[0] : `nothing installed, for these reasons:\n${lines.map((l) => ` - ${l}`).join("\n")}`}${lines.length > 1 ? tail : ""}` };
|
|
1079
|
+
}
|
|
1080
|
+
return { foreign };
|
|
1081
|
+
}
|
|
1082
|
+
|
|
1083
|
+
async function installLinuxUser(ctx, paths, stack = null, foreign = []) {
|
|
744
1084
|
if (ctx.fs.existsSync(paths.installPath) && !ctx.force) {
|
|
745
1085
|
return fail(ctx.err, `${paths.installPath} already exists — pass --force to replace it (same non-clobber contract as init)`);
|
|
746
1086
|
}
|
|
747
|
-
|
|
748
|
-
|
|
1087
|
+
// Issue #464: every file this run writes, recorded before it is written, so a failure puts them back rather than
|
|
1088
|
+
// leaving half an install (seen with a root-owned ~/.config: the stack's files written, the worker unit refused). All
|
|
1089
|
+
// files are written before any command runs (the worker unit inside applyStack's `beforeRuns`), which is what makes a
|
|
1090
|
+
// write failure all-or-nothing. A COMMAND that fails leaves the stack's files, which units already started from, and
|
|
1091
|
+
// says exactly which files this run wrote.
|
|
1092
|
+
const journal = [];
|
|
1093
|
+
const writeUnit = () => {
|
|
1094
|
+
try {
|
|
1095
|
+
ctx.fs.mkdirSync(dirname(paths.installPath), { recursive: true });
|
|
1096
|
+
journalWrite(ctx.fs, journal, paths.installPath, renderLinuxUnit(ctx, stack?.plan.start ?? [], stack?.venues ?? null));
|
|
1097
|
+
return { ok: true };
|
|
1098
|
+
} catch (err) {
|
|
1099
|
+
return { ok: false, failed: `write ${paths.installPath}`, code: null, message: err?.message };
|
|
1100
|
+
}
|
|
1101
|
+
};
|
|
1102
|
+
const writtenList = () => [...new Set(journal.map((e) => e.path))].join(", ") || "none";
|
|
1103
|
+
if (stack?.pendingPassword) {
|
|
1104
|
+
// Issue #468: the new Valkey password into .env first, journalled like every other file of this run, so a write
|
|
1105
|
+
// that fails later puts .env back too. Never over a value (the writer's never-clobber), the file narrowed to this
|
|
1106
|
+
// account alone, the value never shown.
|
|
1107
|
+
const wrote = writeValkeyPassword(ctx, stack.pendingPassword, journal);
|
|
1108
|
+
if (wrote.error) return fail(ctx.err, `${wrote.error}, so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}`);
|
|
1109
|
+
ctx.out(`generated ${VALKEY_PASSWORD_KEY} into ${wrote.path} (32 random bytes, hex; the value is not shown; the file is now readable by this account only)\n`);
|
|
1110
|
+
}
|
|
1111
|
+
if (stack) {
|
|
1112
|
+
for (const note of stack.notes) ctx.out(`note: ${note}\n`);
|
|
1113
|
+
for (const f of foreign) {
|
|
1114
|
+
ctx.out(`⚠ --force: ${f.container} is not managed by ${f.unit} and will be REPLACED by it${f.unit === QUADLET_FILES.proxy.unit ? "; every job running right now loses the per-job network it had on that proxy" : ""}\n`);
|
|
1115
|
+
}
|
|
1116
|
+
const restartWarning = proxyRestartWarning(stack.plan);
|
|
1117
|
+
if (restartWarning) ctx.out(`⚠ ${restartWarning}\n`);
|
|
1118
|
+
const passwordWarning = valkeyPasswordRestartWarning(stack.plan);
|
|
1119
|
+
if (passwordWarning) ctx.out(`⚠ ${passwordWarning}\n`);
|
|
1120
|
+
if (stack.plan.actions.length > 0) {
|
|
1121
|
+
// Shown, then done: the same lines `up` asks consent for, carried out by the same function.
|
|
1122
|
+
ctx.out(`podman venue (PI_BACKENDS in .env): its stack, as Quadlet units in ${stack.plan.dir}:\n`);
|
|
1123
|
+
for (const action of stack.plan.actions) ctx.out(` ${describeAction(action)}\n`);
|
|
1124
|
+
const applied = await applyStack(stack.plan, { fs: ctx.fs, run: (cmd, args) => run(ctx, cmd, args), journal, beforeRuns: writeUnit });
|
|
1125
|
+
if (!applied.ok) {
|
|
1126
|
+
const what = `${applied.failed} failed (${applied.code === null ? applied.message ?? "command not found" : `exit ${applied.code}`})`;
|
|
1127
|
+
if (!applied.ran) {
|
|
1128
|
+
return fail(ctx.err, `${what}, so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}. Fix it, then re-run this install`);
|
|
1129
|
+
}
|
|
1130
|
+
// A command failed: the stack's units may be running from their files, so only the worker unit is put back,
|
|
1131
|
+
// and the manager is reloaded so it forgets it (best effort; the unit was never enabled or started).
|
|
1132
|
+
const unitOnly = journal.filter((e) => e.path === paths.installPath);
|
|
1133
|
+
const rolled = rollBackWrites(ctx.fs, unitOnly);
|
|
1134
|
+
if (rolled.left.length === 0) await run(ctx, "systemctl", ["--user", "daemon-reload"]);
|
|
1135
|
+
const remain = [...new Set(journal.filter((e) => e.path !== paths.installPath).map((e) => e.path))];
|
|
1136
|
+
return fail(
|
|
1137
|
+
ctx.err,
|
|
1138
|
+
`${what}, so the worker unit was NOT installed: it would only start against a missing queue or proxy (${describeRollBack(rolled, { partial: true })}). The stack files this run wrote remain, since its units run from them: ${remain.join(", ") || "none"}. \`pi-dispatch service uninstall\` removes them if you will not retry. \`systemctl --user status ${stack.plan.start.join(" ")}\` and \`journalctl --user -u <unit>\` have the details; fix it, then re-run this install`,
|
|
1139
|
+
);
|
|
1140
|
+
}
|
|
1141
|
+
}
|
|
1142
|
+
}
|
|
1143
|
+
if (!journal.some((e) => e.path === paths.installPath)) {
|
|
1144
|
+
const wrote = writeUnit();
|
|
1145
|
+
if (!wrote.ok) return fail(ctx.err, `${wrote.failed} failed (${wrote.message}), so nothing was installed: ${describeRollBack(rollBackWrites(ctx.fs, journal))}. Fix it, then re-run this install`);
|
|
1146
|
+
}
|
|
749
1147
|
const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
|
|
750
|
-
if (reload === null) return fail(ctx.err, `systemctl not found
|
|
1148
|
+
if (reload === null) return fail(ctx.err, `systemctl not found: is this a systemd host? The unit is written at ${paths.installPath} (files this run wrote, which remain: ${writtenList()})`);
|
|
751
1149
|
const enable = await run(ctx, "systemctl", ["--user", "enable", "--now", paths.name]);
|
|
752
1150
|
if (enable !== 0) {
|
|
753
|
-
return fail(ctx.err, `systemctl --user enable --now ${paths.name} failed (exit ${enable})
|
|
1151
|
+
return fail(ctx.err, `systemctl --user enable --now ${paths.name} failed (exit ${enable}); the unit is written at ${paths.installPath}; \`systemctl --user status ${paths.name}\` has the details (files this run wrote, which remain: ${writtenList()})`);
|
|
754
1152
|
}
|
|
755
1153
|
ctx.out(`installed ${paths.name} → ${paths.installPath} (enabled and started in your user manager)\n`);
|
|
1154
|
+
if (stack) {
|
|
1155
|
+
// Said as it happened: `start` on a unit already active is a no-op, so a unit whose file this run replaced was
|
|
1156
|
+
// RESTARTED (planStack), and the line names which is which rather than calling everything "started".
|
|
1157
|
+
const restarted = stack.plan.restart ?? [];
|
|
1158
|
+
const started = stack.plan.start.filter((u) => !restarted.includes(u));
|
|
1159
|
+
if (started.length > 0) ctx.out(`started ${started.join(" ")} (Quadlet units: never enabled, the generator reads their own [Install] section; a unit already running is left as it is)\n`);
|
|
1160
|
+
// Named by the file that changed (round 3 nit): the proxy also restarts for its rules copy alone.
|
|
1161
|
+
for (const unit of restarted) {
|
|
1162
|
+
// Issue #468: a file this install wrote NEW for a unit it restarts (the Valkey's first password file) is named too.
|
|
1163
|
+
const changedFiles = stack.plan.files.filter((f) => f.state !== "same" && f.restarts === unit).map((f) => f.path);
|
|
1164
|
+
// A unit restarted with its files unchanged (a keeper that did not hold, the proxy after it) says so rather than
|
|
1165
|
+
// ending on "replaced " with nothing after it, as it did.
|
|
1166
|
+
ctx.out(`restarted ${unit}, ${changedFiles.length > 0 ? `because this install replaced ${changedFiles.join(" and ")}` : "as the plan above says (its files are unchanged)"}\n`);
|
|
1167
|
+
}
|
|
1168
|
+
if ((stack.plan.clientsRestarted ?? []).length > 0) ctx.out(`restarted ${stack.plan.clientsRestarted.join(" ")} after it, so they send the new ${VALKEY_PASSWORD_KEY}\n`);
|
|
1169
|
+
if (stack.plan.start.length > 0) ctx.out(`${paths.name} Wants= and is After= ${stack.plan.start.join(" ")}\n`);
|
|
1170
|
+
// PR #463 round 3: a keeper started or restarted beside a proxy this install does not own (PI_EGRESS_PROXY naming
|
|
1171
|
+
// the operator's own) leaves that proxy up since before the keeper; said, with its name.
|
|
1172
|
+
const hint = keeperUnderRunningProxyHint(stack.plan, stack.proxy);
|
|
1173
|
+
if (hint) ctx.out(`⚠ ${hint}\n`);
|
|
1174
|
+
// A WARNING, not a refusal, like the note below it: a worker that runs only while its operator is logged in is a
|
|
1175
|
+
// real (desktop) deployment. But on this venue linger is also what brings the queue and the proxy back, and it
|
|
1176
|
+
// was measured both ways, so the line says which of the two this host is rather than the general caution.
|
|
1177
|
+
ctx.out(lingerNote(await readLinger(ctx.user, (cmd, args) => runCapture(ctx, cmd, args)), ctx.user));
|
|
1178
|
+
return 0;
|
|
1179
|
+
}
|
|
756
1180
|
// Without linger a user manager only runs while a session exists — fine on a desktop, a silent
|
|
757
1181
|
// no-worker-after-reboot on a headless box. Say so instead of letting the operator find out.
|
|
758
1182
|
ctx.out(`note: user units run while you have a session. For a headless host that must start at boot: sudo loginctl enable-linger ${ctx.user}\n`);
|
|
759
1183
|
return 0;
|
|
760
1184
|
}
|
|
761
1185
|
|
|
1186
|
+
/**
|
|
1187
|
+
* Write a new VALKEY_PASSWORD into the deployment's `.env` (issue #468) through the project's one `.env` writer, never
|
|
1188
|
+
* over a value, the file narrowed to this account alone (`narrow`). Recorded in `journal` with the bytes that were there,
|
|
1189
|
+
* so a later failure of the same install puts it back. `{ path }` or `{ error }`; the value is never in either.
|
|
1190
|
+
*/
|
|
1191
|
+
function writeValkeyPassword(ctx, password, journal) {
|
|
1192
|
+
const envPath = join(ctx.deployDir, ".env");
|
|
1193
|
+
let previous;
|
|
1194
|
+
try {
|
|
1195
|
+
previous = ctx.fs.readFileSync(envPath);
|
|
1196
|
+
} catch (err) {
|
|
1197
|
+
return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: it could not be read (${err?.message ?? err})` };
|
|
1198
|
+
}
|
|
1199
|
+
try {
|
|
1200
|
+
const res = updateEnvFile(envPath, VALKEY_PASSWORD_KEY, password, { fs: ctx.fs, platform: ctx.platform, narrow: true });
|
|
1201
|
+
if (!res.changed) return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: the file already assigns it (set while this install ran?). Re-run the install, which then uses that value` };
|
|
1202
|
+
} catch (err) {
|
|
1203
|
+
return { error: `${VALKEY_PASSWORD_KEY} could not be written into ${envPath}: ${err?.message ?? err}` };
|
|
1204
|
+
}
|
|
1205
|
+
journal.push({ path: envPath, existed: true, previous });
|
|
1206
|
+
return { path: envPath };
|
|
1207
|
+
}
|
|
1208
|
+
|
|
762
1209
|
async function installLinuxSystem(ctx, paths) {
|
|
763
1210
|
if (ctx.fs.existsSync(paths.installPath) && !ctx.force) {
|
|
764
1211
|
return fail(ctx.err, `${paths.installPath} already exists — pass --force to re-stage the render (the printed sudo commands would overwrite it)`);
|
|
@@ -815,6 +1262,20 @@ async function doUninstall(ctx) {
|
|
|
815
1262
|
return 0;
|
|
816
1263
|
}
|
|
817
1264
|
const userPath = ctx.platform === "darwin" ? paths.installPath : paths.userPath;
|
|
1265
|
+
// The podman venue's Quadlet units (issue #430), removed with the worker, or on their own when `up` installed them
|
|
1266
|
+
// and no worker unit was ever written. Decided by what EXISTS, not by today's PI_BACKENDS: an operator who dropped
|
|
1267
|
+
// podman from the list still has the units, and they are still this tool's to remove.
|
|
1268
|
+
const quadlets = ctx.platform === "linux" && ctx.which === "worker" && ctx.scope === "user" ? ALL_QUADLET_FILES.filter((q) => ctx.fs.existsSync(join(quadletDir(ctx.home), q.file))) : [];
|
|
1269
|
+
// No user manager to talk to (`sudo -iu`, round 3 D3, measured): every `systemctl --user` below would fail, and the
|
|
1270
|
+
// old code ignored the exits, printed success with exit 0, and left the worker and the stack running with a
|
|
1271
|
+
// dangling default.target.wants link. Refused before anything is touched, whenever this run would talk to it.
|
|
1272
|
+
if (ctx.platform === "linux" && (ctx.fs.existsSync(userPath) || quadlets.length > 0)) {
|
|
1273
|
+
const bus = userBusRefusal({ env: ctx.env, user: ctx.user, euid: ctx.euid });
|
|
1274
|
+
if (bus) return fail(ctx.err, bus.replace("so `systemctl --user` would fail after the files were written", "so `systemctl --user` could stop and disable nothing"));
|
|
1275
|
+
}
|
|
1276
|
+
if (!ctx.fs.existsSync(userPath) && quadlets.length > 0 && !ctx.fs.existsSync(paths.systemPath)) {
|
|
1277
|
+
return removeQuadlets(ctx, quadlets);
|
|
1278
|
+
}
|
|
818
1279
|
if (!ctx.fs.existsSync(userPath)) {
|
|
819
1280
|
// Say where it looked — both scopes — and if the unit turns out to live in ROOT scope, print
|
|
820
1281
|
// the removal commands instead of touching them (the same never-root doctrine as install).
|
|
@@ -833,13 +1294,94 @@ async function doUninstall(ctx) {
|
|
|
833
1294
|
ctx.out(`uninstalled ${paths.name} (booted out of gui/${ctx.euid}, plist removed)\n`);
|
|
834
1295
|
return 0;
|
|
835
1296
|
}
|
|
836
|
-
|
|
1297
|
+
// A non-zero exit here is a unit that is still enabled or still running (round 3, D3): nothing is removed and nothing
|
|
1298
|
+
// is claimed. `disable --now` on an installed unit that is merely not enabled exits 0.
|
|
1299
|
+
const disabled = await run(ctx, "systemctl", ["--user", "disable", "--now", paths.name]);
|
|
1300
|
+
if (disabled !== 0) {
|
|
1301
|
+
return fail(ctx.err, `systemctl --user disable --now ${paths.name} failed (${disabled === null ? "systemctl not found" : `exit ${disabled}`}), so nothing was removed: the worker may still be running. \`systemctl --user status ${paths.name}\` has the details`);
|
|
1302
|
+
}
|
|
837
1303
|
ctx.fs.unlinkSync(userPath);
|
|
838
|
-
await run(ctx, "systemctl", ["--user", "daemon-reload"]);
|
|
1304
|
+
const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
|
|
1305
|
+
if (reload !== 0) {
|
|
1306
|
+
return fail(ctx.err, `${paths.name} is disabled, stopped and its unit file removed, but systemctl --user daemon-reload failed (${reload === null ? "systemctl not found" : `exit ${reload}`}): run it yourself`);
|
|
1307
|
+
}
|
|
839
1308
|
ctx.out(`uninstalled ${paths.name} (disabled, stopped, unit removed)\n`);
|
|
1309
|
+
if (quadlets.length > 0) return removeQuadlets(ctx, quadlets);
|
|
840
1310
|
return 0;
|
|
841
1311
|
}
|
|
842
1312
|
|
|
1313
|
+
/**
|
|
1314
|
+
* Stop and remove the podman venue's Quadlet units. `stop`, never `disable`: a generated unit cannot be disabled any
|
|
1315
|
+
* more than enabled, and removing its file plus a daemon-reload is what unlinks it from default.target. The Valkey
|
|
1316
|
+
* volume and the three networks are deliberately LEFT: the volume holds the queue's wait-list, and a re-install that
|
|
1317
|
+
* found it gone would have dropped every waiting job on an uninstall nobody meant as a purge.
|
|
1318
|
+
*/
|
|
1319
|
+
async function removeQuadlets(ctx, quadlets) {
|
|
1320
|
+
// The network units too (round 3, D2, measured): left behind, they stay `active (exited)` in the manager, so after the
|
|
1321
|
+
// `podman network rm` the message below suggests, a reinstall in the same manager lifetime found its network unit
|
|
1322
|
+
// already "started" and the container failed with "network not found". Containers first, then their networks.
|
|
1323
|
+
const containers = quadlets.filter((q) => q.file.endsWith(".container")).map((q) => q.unit);
|
|
1324
|
+
const networks = quadlets.filter((q) => q.file.endsWith(".network")).map((q) => q.unit);
|
|
1325
|
+
const units = [...containers, ...networks];
|
|
1326
|
+
if (units.length > 0) {
|
|
1327
|
+
// A unit that is already stopped stops with exit 0; a non-zero exit is a stop that did not happen (no manager, a
|
|
1328
|
+
// unit that would not die), and removing the files under a running container would leave it running unmanaged
|
|
1329
|
+
// while this command claimed it was gone (round 3, D3).
|
|
1330
|
+
const stopped = await run(ctx, "systemctl", ["--user", "stop", ...units]);
|
|
1331
|
+
// Issue #464: a unit the manager never loaded (an install or `up` whose daemon-reload failed after writing the
|
|
1332
|
+
// files) makes `stop` exit 5 ("not loaded"), and uninstall then could not remove the very files that failed run
|
|
1333
|
+
// left. So a failed stop is asked about unit by unit: one that is not loaded, or loaded and not running, is
|
|
1334
|
+
// stopped as far as it can be; only a unit still active refuses the removal, as before.
|
|
1335
|
+
if (stopped !== 0 && stopped !== null && (await unitsStillRunning(ctx, units)).length === 0) {
|
|
1336
|
+
ctx.out(`note: systemctl --user stop exited ${stopped}, and none of ${units.join(", ")} is running (not loaded, or already stopped), so their files are removed\n`);
|
|
1337
|
+
} else if (stopped !== 0) {
|
|
1338
|
+
return fail(ctx.err, `systemctl --user stop ${units.join(" ")} failed (${stopped === null ? "systemctl not found" : `exit ${stopped}`}), so nothing was removed: the containers may still be running. \`systemctl --user status ${units.join(" ")}\` has the details`);
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
for (const q of quadlets) ctx.fs.unlinkSync(join(quadletDir(ctx.home), q.file));
|
|
1342
|
+
// The account-owned copy of the proxy's rules (E1) goes with its unit, and the Valkey's password file with its unit
|
|
1343
|
+
// (issue #468; the password itself stays in the deployment's .env, where a re-install finds it).
|
|
1344
|
+
const conf = proxyConfCopyPath(ctx.home);
|
|
1345
|
+
if (ctx.fs.existsSync(conf)) ctx.fs.unlinkSync(conf);
|
|
1346
|
+
const valkeyEnv = valkeyEnvPath(ctx.home);
|
|
1347
|
+
if (quadlets.some((q) => q.file === QUADLET_FILES.valkey.file) && ctx.fs.existsSync(valkeyEnv)) ctx.fs.unlinkSync(valkeyEnv);
|
|
1348
|
+
const reload = await run(ctx, "systemctl", ["--user", "daemon-reload"]);
|
|
1349
|
+
if (reload !== 0) {
|
|
1350
|
+
return fail(ctx.err, `the Quadlet files are removed and their units stopped, but systemctl --user daemon-reload failed (${reload === null ? "systemctl not found" : `exit ${reload}`}): run it yourself so the manager forgets them`);
|
|
1351
|
+
}
|
|
1352
|
+
// squid ignores SIGTERM, so its stop takes systemd's 10 s and ends in SIGKILL, exit 137, and the unit is left
|
|
1353
|
+
// `failed` (measured): after the file is gone `systemctl --user list-units` would show a not-found failed unit
|
|
1354
|
+
// forever. reset-failed clears it. Its exit is not checked: a unit that never failed is already what it asks for.
|
|
1355
|
+
if (units.length > 0) await run(ctx, "systemctl", ["--user", "reset-failed", ...units]);
|
|
1356
|
+
// The volume is named only when a Valkey unit was among what was removed (issue #452 gate round 2): a stack installed
|
|
1357
|
+
// without Valkey never had one, and the line used to claim a volume nothing here made was kept.
|
|
1358
|
+
const hadValkey = quadlets.some((q) => q.file === QUADLET_FILES.valkey.file);
|
|
1359
|
+
ctx.out(`removed the podman venue's Quadlet units (${quadlets.map((q) => q.file).join(", ")}); ${hadValkey ? "the pi-dispatch-valkey-data volume and the networks are" : "the networks are"} kept, remove them with podman if you mean to\n`);
|
|
1360
|
+
return 0;
|
|
1361
|
+
}
|
|
1362
|
+
|
|
1363
|
+
/**
|
|
1364
|
+
* The units of `units` the user manager still has running, asked one by one (`systemctl --user show`), for an
|
|
1365
|
+
* uninstall whose `stop` failed (issue #464). A unit is taken as not running only on the manager's own answer: not
|
|
1366
|
+
* loaded (`LoadState=not-found`), or loaded and `inactive` or `failed`. A unit it could not be asked about counts as
|
|
1367
|
+
* running, so the removal is refused rather than guessed.
|
|
1368
|
+
*/
|
|
1369
|
+
async function unitsStillRunning(ctx, units) {
|
|
1370
|
+
const running = [];
|
|
1371
|
+
for (const unit of units) {
|
|
1372
|
+
const shown = await runQuery(ctx, "systemctl", ["--user", "show", "--property=LoadState,ActiveState", unit]);
|
|
1373
|
+
const props = Object.fromEntries(
|
|
1374
|
+
String(shown.stdout ?? "")
|
|
1375
|
+
.split("\n")
|
|
1376
|
+
.map((l) => l.trim().split("="))
|
|
1377
|
+
.filter((kv) => kv.length === 2),
|
|
1378
|
+
);
|
|
1379
|
+
const idle = shown.code === 0 && (props.LoadState === "not-found" || (props.LoadState === "loaded" && ["inactive", "failed"].includes(props.ActiveState)));
|
|
1380
|
+
if (!idle) running.push(unit);
|
|
1381
|
+
}
|
|
1382
|
+
return running;
|
|
1383
|
+
}
|
|
1384
|
+
|
|
843
1385
|
/** Informational only — reports every scope it knows about and always exits 0. */
|
|
844
1386
|
async function doStatus(ctx) {
|
|
845
1387
|
const paths = unitPaths(ctx);
|
|
@@ -875,6 +1417,16 @@ async function doStatus(ctx) {
|
|
|
875
1417
|
const systemHits = [paths.systemPath, ...(ctx.which === "worker" ? ["/etc/systemd/system/worker.service"] : [])].filter((p) => ctx.fs.existsSync(p));
|
|
876
1418
|
ctx.out(systemHits.length ? `system scope: ${systemHits.join(", ")} EXISTS — not managed by this tool\n` : "system scope: none\n");
|
|
877
1419
|
reportEnvSetup(ctx, [paths.userPath, ...systemHits]);
|
|
1420
|
+
// The podman venue's Quadlet units, one line each, and nothing at all where there are none, so the output of every
|
|
1421
|
+
// other deployment is byte-identical to before issue #430.
|
|
1422
|
+
if (ctx.which === "worker") {
|
|
1423
|
+
for (const q of ALL_QUADLET_FILES) {
|
|
1424
|
+
const path = join(quadletDir(ctx.home), q.file);
|
|
1425
|
+
if (!ctx.fs.existsSync(path)) continue;
|
|
1426
|
+
const active = await runCapture(ctx, "systemctl", ["--user", "is-active", q.unit]);
|
|
1427
|
+
ctx.out(`quadlet: ${path} (${q.unit}): ${active.code === null ? "systemctl not found" : active.output.trim() || "unknown"}\n`);
|
|
1428
|
+
}
|
|
1429
|
+
}
|
|
878
1430
|
return 0;
|
|
879
1431
|
}
|
|
880
1432
|
|
|
@@ -945,13 +1497,15 @@ async function doRestart(ctx, values) {
|
|
|
945
1497
|
return fail(ctx.err, `--drain-timeout must be a positive number of seconds, got: ${values["drain-timeout"]}`);
|
|
946
1498
|
}
|
|
947
1499
|
let queue = ctx.queue;
|
|
1500
|
+
let url;
|
|
948
1501
|
if (!queue) {
|
|
949
1502
|
// VALKEY_URL only, exactly like cli.mjs's pause/resume: the drain must work even when the rest
|
|
950
1503
|
// of the config (forge auth …) is broken, and failFast keeps a down Valkey an error in seconds
|
|
951
1504
|
// instead of a hung restart. Lazy imports for the same reason cli.mjs uses them: `service`
|
|
952
1505
|
// subcommands that never touch the queue must not load bullmq/ioredis.
|
|
953
|
-
|
|
954
|
-
const { parseConnection } = await import("./connection.mjs");
|
|
1506
|
+
// This shell's VALKEY_URL, else the deployment .env's (PR #475's review), as every CLI verb reads it.
|
|
1507
|
+
const { cliValkeyUrl, parseConnection } = await import("./connection.mjs");
|
|
1508
|
+
url = cliValkeyUrl(ctx.env, { cwd: ctx.deployDir, warn: (line) => ctx.err(line) });
|
|
955
1509
|
const { makeQueue } = await import("./queue.mjs");
|
|
956
1510
|
queue = makeQueue(parseConnection(url, { failFast: true }));
|
|
957
1511
|
}
|
|
@@ -981,7 +1535,7 @@ async function doRestart(ctx, values) {
|
|
|
981
1535
|
// this command would otherwise be doing: reporting a drained queue it cannot see all of.
|
|
982
1536
|
const delayed = await queue.getJobCounts("delayed").then((c) => Number(c?.delayed ?? 0), () => 0);
|
|
983
1537
|
if (delayed > 0) {
|
|
984
|
-
ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
|
|
1538
|
+
ctx.out(`note: ${delayed} job(s) sit in the delayed set (cron next-occurrences, retry backoff, quiet hours, scope/host deferrals, or jobs held on run.waitFor). None is active, so none blocked this drain; they will wake against the new version.\n`);
|
|
985
1539
|
}
|
|
986
1540
|
const stopped = await doStop(ctx);
|
|
987
1541
|
if (stopped !== 0) {
|
|
@@ -997,7 +1551,7 @@ async function doRestart(ctx, values) {
|
|
|
997
1551
|
ctx.out("resumed — drained restart complete\n");
|
|
998
1552
|
return 0;
|
|
999
1553
|
} catch (error) {
|
|
1000
|
-
return fail(ctx.err, `could not reach Valkey
|
|
1554
|
+
return fail(ctx.err, `could not reach Valkey: ${(await import("./valkey-auth.mjs")).valkeyDownHint(url)}\n ${error.message}`);
|
|
1001
1555
|
} finally {
|
|
1002
1556
|
await queue.close().catch(() => {});
|
|
1003
1557
|
}
|
|
@@ -1013,6 +1567,20 @@ function quoteArgs(args) {
|
|
|
1013
1567
|
return args.map((a) => (a.includes(" ") || a.includes("\\") ? `"${a}"` : a)).join(" ");
|
|
1014
1568
|
}
|
|
1015
1569
|
|
|
1570
|
+
/** Is anything listening on host:port? up.mjs's probe, for up.mjs's reason: a plain connect, no protocol. */
|
|
1571
|
+
function defaultProbeTcp(host, port, timeoutMs = 1500) {
|
|
1572
|
+
return new Promise((resolvePromise) => {
|
|
1573
|
+
const socket = netConnect({ host, port });
|
|
1574
|
+
const done = (result) => {
|
|
1575
|
+
socket.destroy();
|
|
1576
|
+
resolvePromise(result);
|
|
1577
|
+
};
|
|
1578
|
+
socket.setTimeout(timeoutMs, () => done(false));
|
|
1579
|
+
socket.on("connect", () => done(true));
|
|
1580
|
+
socket.on("error", () => done(false));
|
|
1581
|
+
});
|
|
1582
|
+
}
|
|
1583
|
+
|
|
1016
1584
|
/** Exit code of a spawned command; null when it could not launch (not on PATH) — the up.mjs pattern. */
|
|
1017
1585
|
function run(ctx, cmd, args) {
|
|
1018
1586
|
return new Promise((resolvePromise) => {
|
|
@@ -1028,6 +1596,28 @@ function run(ctx, cmd, args) {
|
|
|
1028
1596
|
});
|
|
1029
1597
|
}
|
|
1030
1598
|
|
|
1599
|
+
/**
|
|
1600
|
+
* Like runCapture() with the two streams SEPARATE, for a query whose answer is stdout alone (round 2, E2): podman
|
|
1601
|
+
* prints warnings on stderr on ordinary accounts, and merged into the answer they made a label read as someone else's.
|
|
1602
|
+
*/
|
|
1603
|
+
function runQuery(ctx, cmd, args) {
|
|
1604
|
+
return new Promise((resolvePromise) => {
|
|
1605
|
+
let child;
|
|
1606
|
+
try {
|
|
1607
|
+
child = ctx.spawn(cmd, args, { stdio: ["ignore", "pipe", "pipe"] });
|
|
1608
|
+
} catch {
|
|
1609
|
+
resolvePromise({ code: null, stdout: "", stderr: "" });
|
|
1610
|
+
return;
|
|
1611
|
+
}
|
|
1612
|
+
let stdout = "";
|
|
1613
|
+
let stderr = "";
|
|
1614
|
+
child.stdout?.on("data", (d) => (stdout += d));
|
|
1615
|
+
child.stderr?.on("data", (d) => (stderr += d));
|
|
1616
|
+
child.on("error", () => resolvePromise({ code: null, stdout, stderr }));
|
|
1617
|
+
child.on("close", (code) => resolvePromise({ code, stdout, stderr }));
|
|
1618
|
+
});
|
|
1619
|
+
}
|
|
1620
|
+
|
|
1031
1621
|
/** Like run() but with stdout+stderr captured, for read-only lookups (nssm status, is-active …). */
|
|
1032
1622
|
function runCapture(ctx, cmd, args) {
|
|
1033
1623
|
return new Promise((resolvePromise) => {
|