@edgehero/pi-dispatch 1.10.3 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.example +303 -150
  2. package/README.md +52 -0
  3. package/deploy/com.pi-dispatch.worker.plist +10 -4
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +12 -1
  15. package/deploy/worker-env-wrapper.sh +63 -37
  16. package/deploy/worker.service +18 -8
  17. package/package.json +15 -5
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4756 -394
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +456 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +245 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-pi.mjs +19 -3
  54. package/src/host-registry.mjs +29 -2
  55. package/src/identity.mjs +29 -4
  56. package/src/image-preflight.mjs +46 -11
  57. package/src/image-ref.mjs +21 -0
  58. package/src/index.mjs +363 -13
  59. package/src/init.mjs +197 -38
  60. package/src/job-user.mjs +252 -0
  61. package/src/json-duplicates.mjs +204 -0
  62. package/src/live-probes.mjs +1020 -0
  63. package/src/materialize.mjs +4 -11
  64. package/src/netns-keeper.mjs +264 -0
  65. package/src/on-failure.mjs +119 -0
  66. package/src/outbox.mjs +7 -0
  67. package/src/packages.mjs +2 -2
  68. package/src/podman-stack.mjs +1304 -0
  69. package/src/prepare-github.mjs +6 -6
  70. package/src/prepare-local.mjs +51 -17
  71. package/src/prepare.mjs +27 -6
  72. package/src/pricing.mjs +9 -5
  73. package/src/processor.mjs +506 -26
  74. package/src/provider-key.mjs +66 -0
  75. package/src/provider-steering.mjs +185 -0
  76. package/src/queue.mjs +35 -8
  77. package/src/redact.mjs +84 -0
  78. package/src/reserved-env.mjs +7 -3
  79. package/src/retention-sweep.mjs +178 -0
  80. package/src/run-container.mjs +181 -14
  81. package/src/run-history.mjs +105 -16
  82. package/src/runtime-observations.mjs +1152 -0
  83. package/src/runtime-settings.mjs +13 -8
  84. package/src/sandbox-cli.mjs +100 -95
  85. package/src/sandbox-store.mjs +612 -45
  86. package/src/sandbox.mjs +1459 -37
  87. package/src/schedules.mjs +16 -3
  88. package/src/secret-profiles.mjs +2 -1
  89. package/src/secrets.mjs +24 -6
  90. package/src/service-env.mjs +247 -0
  91. package/src/service.mjs +618 -28
  92. package/src/session-store.mjs +678 -53
  93. package/src/start.mjs +1348 -326
  94. package/src/subscriptions.mjs +7 -3
  95. package/src/transient.mjs +240 -0
  96. package/src/triggers-file.mjs +71 -15
  97. package/src/triggers.mjs +179 -19
  98. package/src/up.mjs +1399 -85
  99. package/src/valkey-auth.mjs +529 -0
  100. package/src/valkey-endpoint.mjs +367 -0
  101. package/src/watch-closer.mjs +158 -0
@@ -1,3 +1,4 @@
1
+ import { GIT_READ_FLAGS } from "./git-hardening.mjs";
1
2
  import { execFile } from "node:child_process";
2
3
  import { mkdirSync, writeFileSync } from "node:fs";
3
4
  import { dirname, isAbsolute, join, relative } from "node:path";
@@ -343,17 +344,9 @@ export async function materializePiDir({ gitDir, sha, destDir, git = defaultGit
343
344
  }
344
345
 
345
346
  async function defaultGit(gitDir, args, { raw = false, maxBuffer = 16 * 1024 * 1024 } = {}) {
346
- // -c protecting against a hostile repo config: no hooks, no external filters, no pager.
347
- const hardened = [
348
- "-c",
349
- "core.hooksPath=/dev/null",
350
- "-c",
351
- "core.fsmonitor=false",
352
- "--no-pager",
353
- "-C",
354
- gitDir,
355
- ...args,
356
- ];
347
+ // -c protecting against a hostile repo config: no hooks, no external filters, no pager. The flags moved
348
+ // to git-hardening.mjs when a seventh copy of them turned out to be missing one (issue #286's sweep).
349
+ const hardened = [...GIT_READ_FLAGS, "-C", gitDir, ...args];
357
350
  const { stdout } = await exec("git", hardened, {
358
351
  encoding: raw ? "buffer" : "utf8",
359
352
  maxBuffer,
@@ -0,0 +1,264 @@
1
+ /**
2
+ * netns-keeper.mjs -- the rootless network keeper (issue #458) and the network detach gate (issue #452).
3
+ *
4
+ * A LEAF above `daemon-facts.mjs` only, so `egress.mjs` can route every detach through `makeDetachGate` without a
5
+ * cycle (`podman-stack.mjs`, where these first lived, imports `egress.mjs`, and re-exports all of them).
6
+ */
7
+
8
+ import { DAEMON_FACTS_ARGS, DAEMON_FACTS_TIMEOUT_MS, PODMAN_INFO_ARGS, parseDaemonFacts, parsePodmanInfo } from "./daemon-facts.mjs";
9
+
10
+ /**
11
+ * The rootless network keeper's container and network name (issue #458). One idle container on an `--internal`,
12
+ * DNS-disabled network of its own, so this account's shared rootless network helper always has a running bridge member.
13
+ * On Podman 4.9, `podman network disconnect` of the proxy from a job's network tears that helper down under the running
14
+ * proxy whenever no other bridge container runs (v4.9.3 libpod/networking_linux.go counts the disconnecting container as
15
+ * the caller and cleans up at one), and every later egress job then gets 503 until the proxy restarts: measured, and
16
+ * gone with the keeper running (deploy/pi-dispatch-netns-keeper.container has the rest).
17
+ *
18
+ * Outside every prefix a pi-dispatch sweep removes, by a test: a sweep that took it would bring the defect back with
19
+ * nothing said, and a `network rm -f` of its network leaves its unit failed until restarted (measured).
20
+ */
21
+ export const NETNS_KEEPER = "pi-dispatch-netns-keeper";
22
+
23
+ /**
24
+ * How the keeper is read, everywhere it is read (doctor, `up`, the worker's egress preflight): its state, its network
25
+ * MODE and the networks it is on, in one `podman inspect`. Running is not enough (issue #463 gate, measured on 4.9.3):
26
+ * a container of this name on `--network none`, `slirp4netns`, `pasta` or `host` runs and holds nothing open, since only
27
+ * a running member of a BRIDGE network keeps the rootless helper alive. What 4.9.3 printed for each, with this format:
28
+ * `running|bridge|pi-dispatch-netns-keeper,` (the shipped unit), `running|none|none,`, `running|slirp4netns|`,
29
+ * `running|pasta|`, `running|host|host,`, `paused|bridge|pi-dispatch-netns-keeper,`, and exit 125 for no container.
30
+ */
31
+ export const NETNS_KEEPER_FORMAT = "--format={{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $k, $v := .NetworkSettings.Networks}}{{$k}},{{end}}|{{.State.StartedAt.UnixMilli}}";
32
+
33
+ /** When a container started, in epoch milliseconds (`{{.State.StartedAt.UnixMilli}}`, measured on 4.9.3). */
34
+ export const STARTED_AT_FORMAT = "--format={{.State.StartedAt.UnixMilli}}";
35
+
36
+ /**
37
+ * How long the keeper must have been running to count (PR #463 round 2, measured on 4.9.3): a keeper killed every
38
+ * 0.5 s read as running on its bridge in 57 of 224 back-to-back reads, so a crash loop could pass a single read.
39
+ * Restart=always brings a killed keeper back in about 1.25 s, so a keeper in such a loop never reaches this age.
40
+ */
41
+ export const NETNS_KEEPER_MIN_AGE_MS = 3_000;
42
+
43
+ /**
44
+ * How much later than the proxy the keeper may have started and still count (PR #463 round 2). A keeper that started
45
+ * AFTER the proxy was down while the proxy ran, and a job teardown in that gap cuts the proxy's route out for good
46
+ * (measured: a kill and a teardown right after it, then the keeper back and holding, and the proxy still had no route
47
+ * out until IT restarted). Nothing outside shows that damage, so the order is the only evidence there is. The grace is
48
+ * for a start of both together, and no wider (PR #463 round 3): measured on 4.9.3, the keeper started 4 to 52 ms from the
49
+ * proxy in three `service install`s, 17 to 50 ms in three boot-like starts (the user manager restarted), and 63 ms in
50
+ * an `up --yes`. 15 s is two orders of magnitude over that, and it bounds the one window the rule cannot see: a keeper
51
+ * death AND a teardown both within 15 s of a joint start (it was a minute, and a review measured that window used).
52
+ */
53
+ export const NETNS_KEEPER_AFTER_PROXY_GRACE_MS = 15_000;
54
+
55
+ /**
56
+ * A YOUNG KEEPER IS WAITED OUT, NOT FAILED (issue #476). When the stack starts together (a `service install`, a boot
57
+ * with linger) the worker reads the keeper under a second after it started (measured 0.5 to 0.8 s on 4.9.3), so the
58
+ * age rule above failed every such start: a boot warning with a remedy that restarts the keeper after the proxy, and a
59
+ * queued job that spent an attempt. A keeper that is running on its own bridge holds the helper open now; the age rule
60
+ * is there to catch a keeper that keeps dying, which a wait tells apart. So a keeper whose only fault is its age, and
61
+ * that did not start out of order against the proxy, is `young`: the boot waits until it is `NETNS_KEEPER_MIN_AGE_MS`
62
+ * old plus this margin and judges once more, and a job is moved to the delayed set for as long, without an attempt.
63
+ */
64
+ export const NETNS_KEEPER_YOUNG_MARGIN_MS = 1_000;
65
+
66
+ /**
67
+ * How long one job may be held on a young keeper (issue #476). A keeper that stays up needs one wait (at most the
68
+ * minimum age plus the margin, 4 s); a crash loop (Restart=always, RestartSec=1s) is young at every start, and would
69
+ * hold a job for ever. So the hold ends at the first sign of a loop (a new start, or no keeper, while the job waited)
70
+ * and in any case after this bound, with its own reason, `netns-keeper-crash-loop`.
71
+ */
72
+ export const NETNS_KEEPER_YOUNG_HOLD_MAX_MS = 30_000;
73
+
74
+ /** The wait a young keeper needs, from its age: until it is old enough, plus the margin, never less than the margin. */
75
+ export function netnsKeeperYoungWaitMs(ageMs) {
76
+ // A start read in the future (or no age at all) counts as age 0: the whole minimum, and no more.
77
+ const age = Number.isFinite(ageMs) ? Math.max(0, ageMs) : 0;
78
+ return Math.max(0, NETNS_KEEPER_MIN_AGE_MS - age) + NETNS_KEEPER_YOUNG_MARGIN_MS;
79
+ }
80
+
81
+ /**
82
+ * The sentence for a later attempt of a job that already saw the keeper loop (issue #476, gate of PR #479): `seen`, the
83
+ * keeper starts the earlier attempt saw; `problem`, what the read now found, in words that follow the keeper's name.
84
+ * Without it the retry a minute later met a keeper that had restarted out of order against the proxy (or none at all)
85
+ * and failed as `netns-keeper-not-holding`, so the loop's own comment was almost never the one a job ended on.
86
+ */
87
+ export function netnsKeeperLoopAgainSentence({ seen, problem = null, remedy }) {
88
+ const at = (ms) => (Number.isFinite(ms) ? new Date(ms).toISOString() : "an unread time");
89
+ const starts = (Array.isArray(seen) ? seen : []).map(at).join(" and ") || "unread times";
90
+ return `the rootless network keeper ${NETNS_KEEPER} keeps restarting (a crash loop): an earlier attempt of this job saw it start at ${starts}, and now it ${problem ?? "does not hold"}, so it holds nothing open for long, and on Podman 4.x a job network's teardown in a gap would cut the egress proxy's route out (issue #458), so this job is retried rather than started. To fix it, find why it exits (journalctl --user -u ${NETNS_KEEPER}.service -n 50), then ${remedy ?? `start it as the worker's account: systemctl --user restart ${NETNS_KEEPER}.service`}`;
91
+ }
92
+
93
+ /**
94
+ * The sentence for a keeper that kept restarting while a job waited for it (issue #476). `was`: the start the job first
95
+ * waited on; `now`: the start read now, or `null` when no running keeper was read; `problem`: what the read now found,
96
+ * in words that follow the keeper's name, when it found no running keeper; `heldMs`: how long the job had waited.
97
+ */
98
+ export function netnsKeeperCrashLoopSentence({ was, now = null, problem = null, heldMs, remedy }) {
99
+ const at = (ms) => (Number.isFinite(ms) ? new Date(ms).toISOString() : "an unread time");
100
+ const seen = now === null ? `and ${Math.round(heldMs / 100) / 10} s later it ${problem ?? "was not running"}` : now === was ? `and ${Math.round(heldMs / 100) / 10} s later it still read as younger than ${NETNS_KEEPER_MIN_AGE_MS / 1000} s` : `and started again at ${at(now)}`;
101
+ return `the rootless network keeper ${NETNS_KEEPER} keeps restarting (a crash loop): a job waited for the one started at ${at(was)} to have run ${NETNS_KEEPER_MIN_AGE_MS / 1000} s, ${seen}, so it holds nothing open for long, and on Podman 4.x a job network's teardown in a gap would cut the egress proxy's route out (issue #458), so this job is retried rather than started. To fix it, find why it exits (journalctl --user -u ${NETNS_KEEPER}.service -n 50), then ${remedy ?? `start it as the worker's account: systemctl --user restart ${NETNS_KEEPER}.service`}`;
102
+ }
103
+
104
+ /**
105
+ * Judge one read of `podman inspect NETNS_KEEPER_FORMAT NETNS_KEEPER` (`{ code, stdout }`, stdout alone). Returns
106
+ * `{ holds, exists, running, restartProxy, problem }`: `holds` only when the keeper is running, in bridge mode, and
107
+ * attached to its own network, the one shape measured to keep the helper alive; `problem` says, in words that follow
108
+ * the keeper's name, what is wrong otherwise. A container that also sits on other networks still holds (it is a bridge
109
+ * member); only the shipped unit's shape is required, not its exact network list.
110
+ *
111
+ * With `now` (a clock, milliseconds) it also has to have been running for NETNS_KEEPER_MIN_AGE_MS; with
112
+ * `proxyStartedMs` it must not have started more than NETNS_KEEPER_AFTER_PROXY_GRACE_MS after the proxy, and one that
113
+ * did is `restartProxy`: the remedy is then the PROXY's restart, not the keeper's. `up` passes neither: it asks only
114
+ * whether a keeper is there to leave alone. doctor and the worker's preflight pass both.
115
+ *
116
+ * A keeper whose ONLY fault is its age (running on its own bridge, not started out of order) also carries `young`,
117
+ * `{ startedMs, ageMs, waitMs }` (issue #476): it is still not held here, and doctor still says so, but the worker's
118
+ * boot, its per-job preflight and the sandbox opener wait `waitMs` for it rather than fail on it.
119
+ */
120
+ export function judgeNetnsKeeper({ code, stdout }, { now = null, proxyStartedMs = null } = {}) {
121
+ // `thenRestartProxy` (PR #463 round 3): a keeper that does not hold under a proxy already up longer than the grace will,
122
+ // once started, have started after it, and the order rule will then ask for the proxy's restart; so the fix says both
123
+ // steps now rather than one per retry.
124
+ const late = now !== null && Number.isFinite(proxyStartedMs) && now - proxyStartedMs > NETNS_KEEPER_AFTER_PROXY_GRACE_MS;
125
+ const no = (fields) => ({ holds: false, exists: true, running: true, restartProxy: false, thenRestartProxy: late && fields.restartProxy !== true, ...fields });
126
+ if (code !== 0) return no({ exists: false, running: false, problem: "is not under this account's Podman" });
127
+ const [status = "", mode = "", nets = "", started = ""] = String(stdout ?? "").trim().split("|");
128
+ const networks = nets.split(",").filter(Boolean);
129
+ if (status !== "running") return no({ running: false, problem: `is ${status || "not running"} under this account's Podman` });
130
+ if (mode !== "bridge") return no({ problem: `is running on the ${mode || "unreported"} network mode, not on its ${NETNS_KEEPER} bridge network, so it holds nothing open` });
131
+ if (!networks.includes(NETNS_KEEPER)) return no({ problem: `is running but not attached to its ${NETNS_KEEPER} bridge network (it is on ${networks.join(", ") || "none"}), which is not the keeper this project ships` });
132
+ if (now !== null) {
133
+ const startedMs = /^\d+$/.test(started) ? Number(started) : null;
134
+ if (startedMs === null) return no({ problem: "is running, but when it started could not be read, so whether it has stayed up is not known" });
135
+ const age = now - startedMs;
136
+ const outOfOrder = Number.isFinite(proxyStartedMs) && startedMs > proxyStartedMs + NETNS_KEEPER_AFTER_PROXY_GRACE_MS;
137
+ if (age < NETNS_KEEPER_MIN_AGE_MS) {
138
+ // `young` (issue #476): only when age is the whole fault. One started out of order against the proxy is not
139
+ // waited out, since waiting cannot make its start earlier; its words stay what they were.
140
+ return no({
141
+ problem: `has been running for only ${Math.max(0, Math.round(age / 100) / 10)} s, and a keeper that keeps dying reads as running for a moment at a time; it counts once it has run for ${NETNS_KEEPER_MIN_AGE_MS / 1000} s`,
142
+ ...(outOfOrder ? {} : { young: { startedMs, ageMs: age, waitMs: netnsKeeperYoungWaitMs(age) } }),
143
+ });
144
+ }
145
+ if (outOfOrder) {
146
+ return no({ restartProxy: true, problem: `started ${Math.round((startedMs - proxyStartedMs) / 1000)} s after the egress proxy did, so it was down while the proxy ran, and a job teardown in that gap would have cut the proxy's route out for good, which nothing can see from outside` });
147
+ }
148
+ }
149
+ return { holds: true, exists: true, running: true, restartProxy: false, problem: null };
150
+ }
151
+
152
+ /**
153
+ * Whether this Podman needs the keeper to keep the proxy's route out (issue #458): every 4.x, and a version that cannot
154
+ * be read, since the defect is silent and an unread version is not evidence of 5.x. Only doctor's severity reads it;
155
+ * the installer installs the keeper on every version, where on 5.x it is one idle container (measured harmless on 5.8.1).
156
+ */
157
+ export function podmanNeedsNetnsKeeper(version) {
158
+ const major = /^(\d+)\./.exec(String(version ?? "").trim())?.[1];
159
+ return major === undefined || Number(major) < 5;
160
+ }
161
+
162
+ /**
163
+ * The keeper read a DETACH asks (issue #452, gate round 2): its state, its network mode and the networks it is on, and
164
+ * NOT when it started. `NETNS_KEEPER_FORMAT`'s `{{.State.StartedAt.UnixMilli}}` is Podman's own template and is for the
165
+ * age and order rules, which admit a JOB across time; a detach is safe exactly when the keeper holds at that instant.
166
+ * Measured identical through `podman`, through `podman-docker` and through the real docker CLI against the account's
167
+ * Podman API socket, on 4.9.3 and 5.8.1: `running|bridge|pi-dispatch-netns-keeper,`.
168
+ */
169
+ export const NETNS_KEEPER_NOW_FORMAT = "--format={{.State.Status}}|{{.HostConfig.NetworkMode}}|{{range $k, $v := .NetworkSettings.Networks}}{{$k}},{{end}}";
170
+
171
+ /**
172
+ * THE DETACH GATE (issue #452, gate round 3): the one rule every detach of a container from a network goes through, on
173
+ * every venue and for every caller -- a job's and a sandbox's teardown, the boot reaper, the sandbox sweep, doctor's
174
+ * canary and its sweep, the live probes' peers and their sweep. `egress.mjs`'s `detachEndpoints` is the only code in
175
+ * `worker/src` that issues `network disconnect`, and it asks this first.
176
+ *
177
+ * WHY ONE RULE. On a rootless Podman 4.x, disconnecting a RUNNING container while no other container runs on a bridge
178
+ * network tears the account's rootless network namespace down under the running egress proxy (issue #458), and the
179
+ * proxy then has no route out until it restarts. That is the same whoever disconnects it and through whichever CLI:
180
+ * `podman`, Podman's `podman-docker` emulation answering as `docker`, or the real docker CLI on Podman's API socket
181
+ * (all three measured on 4.9.3). Round 2 guarded some callers; the review then found three more that were not. So the
182
+ * decision moved from the callers into the helper they all share.
183
+ *
184
+ * `gate({ running })`: `null` when the detach may go ahead, else a reason token. Only `running: true` asks anything: a
185
+ * stopped container is not in the rootless namespace, so detaching one is not the trigger. Otherwise ONE runtime read and,
186
+ * where it matters, ONE keeper read, MEMOISED for the gate's lifetime, which is one pass of its caller (one boot reap, one
187
+ * sweep, one doctor run, one job's teardown), an unanswered read included: a CLI that hangs costs one bound per pass, not
188
+ * one per network (the review measured three leftovers taking 90 s). Both reads go through the caller's runner with this
189
+ * module's bound (`DETACH_GATE_READ_TIMEOUT_MS`, `DETACH_GATE_READ_MAX_BUFFER`, the facts readers' own).
190
+ *
191
+ * The runtime read is the one that venue already trusts: `podman info --format json` through `parsePodmanInfo` for the
192
+ * `podman` CLI, and `docker info --format={{json .}}` through `parseDaemonFacts` for `docker`, which recognises Podman on
193
+ * both routes (Podman's own shape from the emulation, Docker's shape with Podman's `ProductLicense` from the API).
194
+ * Docker Engine, rootful Podman (no rootless namespace) and Podman 5.x are let through with no keeper read. A rootless
195
+ * Podman 4.x, or a runtime the read could not identify, needs the keeper to hold AT THAT INSTANT (`judgeNetnsKeeper`
196
+ * without the clock: running, bridge mode, on its own network): `keeper-not-holding`, or `runtime-unreadable`.
197
+ */
198
+ export const DETACH_GATE_READ_TIMEOUT_MS = DAEMON_FACTS_TIMEOUT_MS;
199
+ export const DETACH_GATE_READ_MAX_BUFFER = 1024 * 1024;
200
+
201
+ export function makeDetachGate(run, { bin = "docker", readRuntime = null } = {}) {
202
+ let decided = null;
203
+ const opts = { timeoutMs: DETACH_GATE_READ_TIMEOUT_MS, maxBuffer: DETACH_GATE_READ_MAX_BUFFER };
204
+ const ask = async (args) => {
205
+ try {
206
+ return await run(args, opts);
207
+ } catch {
208
+ return { code: null, stdout: "" };
209
+ }
210
+ };
211
+ async function decide() {
212
+ let runtime = null;
213
+ // `readRuntime`: a caller that has ALREADY read the runtime this pass (doctor's one `docker info`, which its job-user
214
+ // section reads too) hands that read in, `async () => { podman, rootless, version } | null`, so a pass still asks the
215
+ // daemon once. Otherwise the gate reads it itself, with its own bound.
216
+ // `undefined` from it (issue #452, gate round 4) means "I have no answer of my own": the gate then reads it itself,
217
+ // as with no `readRuntime`. `null` is an answer, that the runtime could not be read.
218
+ let handed;
219
+ if (typeof readRuntime === "function") {
220
+ try {
221
+ handed = await readRuntime();
222
+ } catch {
223
+ handed = null;
224
+ }
225
+ }
226
+ runtime = handed === undefined ? await readItself() : handed;
227
+ if (runtime && (runtime.podman !== true || runtime.rootless === false || !podmanNeedsNetnsKeeper(runtime.version))) return null;
228
+ const keeper = await ask(["inspect", NETNS_KEEPER_NOW_FORMAT, NETNS_KEEPER]);
229
+ if (judgeNetnsKeeper({ code: keeper?.code ?? null, stdout: keeper?.stdout ?? "" }).holds) return null;
230
+ return runtime ? "keeper-not-holding" : "runtime-unreadable";
231
+ }
232
+ async function readItself() {
233
+ const read = await ask(bin === "podman" ? PODMAN_INFO_ARGS : DAEMON_FACTS_ARGS);
234
+ let runtime = null;
235
+ if (read?.code === 0) {
236
+ if (bin === "podman") {
237
+ const info = parsePodmanInfo(read.stdout);
238
+ if (info) runtime = { podman: true, rootless: info.rootless, version: info.version };
239
+ } else {
240
+ const facts = parseDaemonFacts(read.stdout)?.facts;
241
+ if (facts) runtime = { podman: facts.podman === true, rootless: facts.rootless, version: facts.serverVersion ?? null };
242
+ }
243
+ }
244
+ return runtime;
245
+ }
246
+ return async function gate({ running = true } = {}) {
247
+ if (running !== true) return null;
248
+ decided ??= decide();
249
+ return decided;
250
+ };
251
+ }
252
+
253
+ /** A `makeDaemonFactsReader` answer as the gate's runtime, or `null` when the daemon did not say. */
254
+ export function runtimeFromFacts(answer) {
255
+ if (answer?.answered !== true || !answer.facts) return null;
256
+ return { podman: answer.facts.podman === true, rootless: answer.facts.rootless, version: answer.facts.serverVersion ?? null };
257
+ }
258
+
259
+ /** The gate's token as the clause a line puts after "because". */
260
+ export function detachBlockedSentence(token, bin = "docker") {
261
+ return token === "runtime-unreadable"
262
+ ? `this shell's ${bin} CLI could not say which container runtime it reaches and the rootless network keeper ${NETNS_KEEPER} does not hold, and on a rootless Podman 4.x detaching the running egress proxy from a network cuts its route out (issue #458)`
263
+ : `the rootless network keeper ${NETNS_KEEPER} does not hold under the rootless Podman 4.x this shell's ${bin} CLI reaches, where detaching the running egress proxy from a network cuts its route out (issue #458)`;
264
+ }
@@ -0,0 +1,119 @@
1
+ /**
2
+ * Running the operator's failure hook (issue #288, `PI_ON_FAILURE`).
3
+ *
4
+ * `wait-check.mjs`, reduced further -- that file is itself the resolver reduced, and every difference
5
+ * here is another subtraction with the reason stated:
6
+ *
7
+ * - **ALL THREE stdio streams are ignored, not counted.** A check's byte counts feed a log line an
8
+ * operator debugs a verdict with; a notification hook has no verdict, and nothing anywhere is
9
+ * entitled to read its output. With no pipes at all, the backgrounded-grandchild EOF hazard that
10
+ * forced `exit`-not-`close` next door cannot even arise (the listener is still `exit`, because it
11
+ * is the right event regardless).
12
+ * - **No verdict, no fault count, no profile table.** The hook changes NOTHING about the job: it runs
13
+ * after the outcome is decided, fire and forget, and a hook fault must never flip a result
14
+ * (CONST-RETRY-INFRA-ONLY). Its exit code is logged and unread -- OQ-027's residual, inherited.
15
+ * - **The reason is shape-guarded HERE, the single normalizing door.** `InfraRetry.reason` defaults to
16
+ * the whole message when no token was given, and messages carry things like a library's own words --
17
+ * so anything that does not look like a fixed token (`/^[a-z0-9-]{1,40}$/`) flattens to "infra"
18
+ * before it can become an argv element. Id-only argv is the contract
19
+ * (`INT-ON-FAILURE-HOOK-CONTRACT`), and one guard at the spawn is worth ten at the call sites.
20
+ *
21
+ * What transfers unchanged: the executable resolved at CALL time (an operator who fixes the path
22
+ * mid-day must not stay broken), `spawn` with an argv ARRAY and `shell: false`, the leading-dash argv
23
+ * refusal, SIGTERM then SIGKILL after a grace, a per-invocation timeout with every timer unref'd, and
24
+ * NEVER throwing -- even the log sink is called inside guards.
25
+ */
26
+
27
+ import { realpathSync, statSync } from "node:fs";
28
+ import { spawn } from "node:child_process";
29
+
30
+ /** SIGTERM, then SIGKILL after this. secrets.mjs's grace, for its reason. */
31
+ const KILL_GRACE_MS = 2000;
32
+
33
+ /** A reason that may ride argv: a fixed lowercase token, never a message. */
34
+ const REASON_SHAPE = /^[a-z0-9-]{1,40}$/;
35
+
36
+ /**
37
+ * Build the hook runner. Returns `fire({ jobId, outcome, reason })` -> void: spawns the operator's
38
+ * command as `cmd <jobId> <outcome> <reason> <host>`, logs one `on_failure` line with the pinned key
39
+ * set { jobId, code, detail }, and never throws, rejects, or delays anything.
40
+ */
41
+ export function makeOnFailure({ command, timeoutMs = 10_000, spawnFn = spawn, realExecutablePath = defaultRealExecutablePath, hostEnv = process.env, host = "", log = () => {} }) {
42
+ const note = (jobId, code, detail) => {
43
+ try {
44
+ log("on_failure", { jobId, code, detail });
45
+ } catch {
46
+ // an injected sink that throws must not break the fire-and-forget contract
47
+ }
48
+ };
49
+
50
+ return function fire({ jobId, outcome, reason }) {
51
+ // Resolved at CALL time, not construction (wait-check's rule): realpath + the executable bit,
52
+ // answering the same question spawn will.
53
+ let path;
54
+ try {
55
+ path = realExecutablePath(command);
56
+ } catch {
57
+ path = null;
58
+ }
59
+ if (!path) return note(typeof jobId === "string" ? jobId : null, null, "unresolvable");
60
+
61
+ const safeReason = typeof reason === "string" && REASON_SHAPE.test(reason) ? reason : "infra";
62
+ // Argv must be non-empty id-only strings, and none may start with a dash -- this is argv where a
63
+ // dash parses as a flag, and the option parser is the operator's. `host` may honestly be "".
64
+ const argv = [jobId, outcome, safeReason, host];
65
+ if (argv.slice(0, 3).some((a) => typeof a !== "string" || a === "") || argv.some((a) => typeof a === "string" && a.startsWith("-"))) {
66
+ return note(typeof jobId === "string" ? jobId : null, null, "argv-unusable");
67
+ }
68
+
69
+ let child;
70
+ try {
71
+ child = spawnFn(path, argv, {
72
+ stdio: ["ignore", "ignore", "ignore"],
73
+ env: hostEnv,
74
+ shell: false,
75
+ });
76
+ } catch {
77
+ return note(jobId, null, "spawn");
78
+ }
79
+
80
+ let done = false;
81
+ let killTimer = null;
82
+ let timedOut = false;
83
+ const finish = (code, detail) => {
84
+ if (done) return;
85
+ done = true;
86
+ clearTimeout(timer);
87
+ note(jobId, code, detail);
88
+ };
89
+ // The kill ladder: SIGTERM, and SIGKILL for a child that ignored it. The escalation timer is
90
+ // deliberately not cleared by finish -- it still has to land on a stubborn child -- and both
91
+ // timers are unref'd so a draining worker is never held open by a notification.
92
+ const stop = () => {
93
+ timedOut = true;
94
+ try {
95
+ child.kill();
96
+ } catch {}
97
+ killTimer = setTimeout(() => {
98
+ try {
99
+ child.kill("SIGKILL");
100
+ } catch {}
101
+ }, KILL_GRACE_MS);
102
+ killTimer.unref?.();
103
+ };
104
+ const timer = setTimeout(() => stop(), timeoutMs);
105
+ timer.unref?.();
106
+
107
+ child.on("error", () => finish(null, "spawn"));
108
+ // `exit`, not `close`: no pipes exist to wait on, and the event is right regardless. A death that
109
+ // followed our own SIGTERM is named a timeout, not blamed on the signal it arrived by.
110
+ child.on("exit", (code, sig) => finish(code, timedOut ? "timeout" : sig ? `signal-${sig}` : "exit"));
111
+ };
112
+ }
113
+
114
+ /** The realpath-and-executable probe, secrets.mjs's exactly. */
115
+ function defaultRealExecutablePath(p) {
116
+ const real = realpathSync(p);
117
+ const st = statSync(real);
118
+ return st.isFile() && (st.mode & 0o111) !== 0 ? real : null;
119
+ }
package/src/outbox.mjs CHANGED
@@ -184,6 +184,13 @@ export function makeCollectChain({ queue, enqueue = enqueueLocalJob, readFlowGat
184
184
  // same operator's flows and, without them, would look up a skill that is not there, write a
185
185
  // plausible report and exit 0. It is NOT part of chainedJobId, for the reason stated below.
186
186
  skillsDir: job.data?.skillsDir,
187
+ // #291, INHERITED -- and here the destructive direction is INVERTED from every line above.
188
+ // For image/skillsDir the hazard of dropping the inheritance is a child MISSING a toolchain;
189
+ // here it is the child GAINING tools the parent's trigger took away: a read-only triage
190
+ // parent would chain a child that can edit and run bash, a widening no operator wrote.
191
+ // Off `job.data`, never off `req` -- the request file can neither set nor drop it (explicit
192
+ // property reads only, above), so the agent cannot widen its child by omitting a key.
193
+ excludeTools: job.data?.excludeTools,
187
194
  // `secrets`/`secretsProfile` are deliberately ABSENT, and the two lines above are exactly why
188
195
  // this comment exists: their reasoning reads as though it should apply here too, and it must
189
196
  // not. An image and a skills directory are toolchain; a resolved credential is a capability.
package/src/packages.mjs CHANGED
@@ -61,7 +61,7 @@ export const NPM_NAME_RE = /^(?:@[a-z0-9][a-z0-9._-]*\/)?[a-z0-9][a-z0-9._-]*$/;
61
61
  * would add an edge to the import-pi <-> packages cycle that only survives because both sides use their
62
62
  * bindings at call time.
63
63
  *
64
- * This list is pi's, not ours: at the 0.80.7 pin `collectPackageResources` falls through to exactly these
64
+ * This list is pi's, not ours: at the 0.99.1 pin (as at 0.80.7) `collectPackageResources` falls through to exactly these
65
65
  * four directory names when `readPiManifest` returns null, so a package with no `pi` key and a `skills/`
66
66
  * dir IS a pi package. See host-pi.mjs's PINNED_PI_NEEDLES for the assertion that keeps that true.
67
67
  */
@@ -281,7 +281,7 @@ export function readStageManifest({ globalPiDir, readFile = readFileSync, fileEx
281
281
  * readStageManifest's policy, because the consumers are advisory (doctor's per-trigger flow lines,
282
282
  * and issue #188's topology) and a half-staged tree must degrade to "nothing visible", not a crash.
283
283
  *
284
- * The semantics mirror pi's collectPackageResources at the 0.80.7 pin EXACTLY, because an enumerator
284
+ * The semantics mirror pi's collectPackageResources at the 0.99.1 pin EXACTLY (unchanged since 0.80.7), because an enumerator
285
285
  * that agrees with pi by hand is how doctor comes to report a tier pi then ignores:
286
286
  * - a `pi` manifest object means its `skills` entries are the ONLY sources -- a manifest WITHOUT a
287
287
  * `skills` key contributes NO skills and gets NO convention fallback (readPiManifest short-circuits