@edgehero/pi-dispatch 1.10.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/.env.example +300 -148
  2. package/README.md +50 -0
  3. package/deploy/com.pi-dispatch.worker.plist +9 -3
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +11 -0
  15. package/deploy/worker-env-wrapper.sh +60 -34
  16. package/deploy/worker.service +18 -8
  17. package/package.json +14 -4
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4701 -414
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +455 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +222 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-registry.mjs +32 -5
  54. package/src/identity.mjs +29 -4
  55. package/src/image-preflight.mjs +46 -11
  56. package/src/image-ref.mjs +21 -0
  57. package/src/index.mjs +387 -17
  58. package/src/init.mjs +197 -38
  59. package/src/job-user.mjs +252 -0
  60. package/src/json-duplicates.mjs +204 -0
  61. package/src/live-probes.mjs +1020 -0
  62. package/src/materialize.mjs +4 -11
  63. package/src/netns-keeper.mjs +264 -0
  64. package/src/on-failure.mjs +119 -0
  65. package/src/outbox.mjs +7 -0
  66. package/src/podman-stack.mjs +1304 -0
  67. package/src/prepare-github.mjs +6 -6
  68. package/src/prepare-local.mjs +51 -17
  69. package/src/prepare.mjs +27 -6
  70. package/src/processor.mjs +505 -26
  71. package/src/provider-key.mjs +41 -0
  72. package/src/provider-steering.mjs +144 -0
  73. package/src/queue.mjs +35 -8
  74. package/src/redact.mjs +84 -0
  75. package/src/reserved-env.mjs +7 -3
  76. package/src/retention-sweep.mjs +178 -0
  77. package/src/run-container.mjs +181 -14
  78. package/src/run-history.mjs +105 -16
  79. package/src/runtime-observations.mjs +1152 -0
  80. package/src/runtime-settings.mjs +13 -8
  81. package/src/sandbox-cli.mjs +100 -95
  82. package/src/sandbox-store.mjs +612 -45
  83. package/src/sandbox.mjs +1459 -37
  84. package/src/schedules.mjs +16 -3
  85. package/src/secret-profiles.mjs +2 -1
  86. package/src/secrets.mjs +23 -6
  87. package/src/service-env.mjs +247 -0
  88. package/src/service.mjs +618 -28
  89. package/src/session-store.mjs +678 -53
  90. package/src/start.mjs +1395 -268
  91. package/src/transient.mjs +240 -0
  92. package/src/triggers-file.mjs +71 -15
  93. package/src/triggers.mjs +176 -19
  94. package/src/up.mjs +1399 -85
  95. package/src/valkey-auth.mjs +529 -0
  96. package/src/valkey-endpoint.mjs +367 -0
  97. package/src/watch-closer.mjs +158 -0
package/src/exit-code.mjs CHANGED
@@ -86,3 +86,18 @@ export function decideWait(exitCode) {
86
86
  return { verdict: "hold", fault: true }; // EXIT_INFRA and every unrecognised code
87
87
  }
88
88
  }
89
+
90
+ /**
91
+ * A process-level printer for a promise nobody handled (PR #475's review, round 2): Node prints such a rejection's
92
+ * reason WHOLE, and a Valkey client's error may carry what it sent (connection.mjs scrubs that at the source; this is
93
+ * the second line of defence for anything else). It prints the message alone and exits 1, which is what Node's own
94
+ * default does with an unhandled rejection: infra, restarted by the service manager, never a silent carry-on. Installed
95
+ * by the worker's CLI (every `pi-dispatch` verb, the worker among them) and the receiver's two entry points.
96
+ */
97
+ export function installRejectionPrinter({ proc = process, write = (line) => process.stderr.write(line) } = {}) {
98
+ proc.on("unhandledRejection", (reason) => {
99
+ const message = reason instanceof Error ? reason.message : typeof reason === "string" ? reason : "a non-Error value";
100
+ write(`error: an unhandled rejection: ${message}\n`);
101
+ proc.exit(EXIT_INFRA);
102
+ });
103
+ }
package/src/flow-gate.mjs CHANGED
@@ -1,3 +1,4 @@
1
+ import { GIT_READ_FLAGS } from "./git-hardening.mjs";
1
2
  import { execFile } from "node:child_process";
2
3
  import { promisify } from "node:util";
3
4
 
@@ -74,9 +75,10 @@ export function aiTriggerAllows(buf) {
74
75
  }
75
76
 
76
77
  async function defaultGit(gitDir, args, { raw = false } = {}) {
77
- // hardening flags mirror materialize.mjs defaultGit — keep in sync. No hooks, no fsmonitor, no
78
- // pager, so a hostile repo config cannot run code or corrupt output during a read.
79
- const hardened = ["-c", "core.hooksPath=/dev/null", "-c", "core.fsmonitor=false", "--no-pager", "-C", gitDir, ...args];
78
+ // No hooks, no fsmonitor, no pager, so a hostile repo config cannot run code or corrupt output during
79
+ // a read. This said "mirror materialize.mjs -- keep in sync" until the copy that had NOT stayed in
80
+ // sync was found (issue #286's sweep); it is an import now, so there is nothing left to remember.
81
+ const hardened = [...GIT_READ_FLAGS, "-C", gitDir, ...args];
80
82
  const { stdout } = await exec("git", hardened, {
81
83
  encoding: raw ? "buffer" : "utf8",
82
84
  maxBuffer: 16 * 1024 * 1024,
@@ -24,6 +24,7 @@ import { configError } from "./config.mjs";
24
24
  import { InfraRetry } from "./processor.mjs";
25
25
  import { fetchFailureReason } from "./gitlab-identity.mjs";
26
26
  import { matchesBranch } from "./gitlab-host.mjs";
27
+ import { isDeterminateFetchFailure } from "./transient.mjs";
27
28
 
28
29
  const API_PREFIX = "/api/v1";
29
30
 
@@ -37,6 +38,15 @@ export function makeForgejoHost({ apiUrl, fetchFn = fetch } = {}) {
37
38
  try {
38
39
  res = await fetchFn(`${root}${path}`, { headers: { Authorization: `token ${token}` }, redirect: "error" });
39
40
  } catch (err) {
41
+ // The one condition where the identity modules and this one now agree, because the issue's
42
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
43
+ // that always redirects are the operator's, and retrying them twice before failing tells
44
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
45
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
46
+ // the delivery and posts publicly that the deployment is misconfigured.
47
+ if (isDeterminateFetchFailure(err)) {
48
+ throw configError(`forgejo-host: GET ${path} failed (${fetchFailureReason(err)})`);
49
+ }
40
50
  throw new InfraRetry(`forgejo-host: GET ${path} failed (${fetchFailureReason(err)})`);
41
51
  }
42
52
  if (res.status === 404 && notFound !== undefined) return notFound;
@@ -114,6 +124,15 @@ export function makeForgejoHost({ apiUrl, fetchFn = fetch } = {}) {
114
124
  redirect: "error",
115
125
  });
116
126
  } catch (err) {
127
+ // The one condition where the identity modules and this one now agree, because the issue's
128
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
129
+ // that always redirects are the operator's, and retrying them twice before failing tells
130
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
131
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
132
+ // the delivery and posts publicly that the deployment is misconfigured.
133
+ if (isDeterminateFetchFailure(err)) {
134
+ throw configError(`forgejo-host: POST ${path} failed (${fetchFailureReason(err)})`);
135
+ }
117
136
  throw new InfraRetry(`forgejo-host: POST ${path} failed (${fetchFailureReason(err)})`);
118
137
  }
119
138
  if (!res.ok) {
@@ -22,6 +22,7 @@
22
22
  */
23
23
 
24
24
  import { configError } from "./config.mjs";
25
+ import { isDeterminateFetchFailure, isJsonContentType, isTransientStatus, responseHeaderReader, transientError } from "./transient.mjs";
25
26
  import { fetchFailureReason } from "./gitlab-identity.mjs";
26
27
 
27
28
  const API_PREFIX = "/api/v1";
@@ -47,7 +48,11 @@ export async function resolveForgejoSelfId({ apiUrl, token, botId = null, fetchF
47
48
  try {
48
49
  res = await fetchFn(url, { headers: { Authorization: `token ${token}` }, redirect: "error" });
49
50
  } catch (err) {
50
- throw configError(`could not resolve the forgejo bot identity from ${url}: ${fetchFailureReason(err)}`);
51
+ // Transient, EXCEPT for the trust and redirect faults below, which is where this deliberately
52
+ // diverges from forgejo-host.mjs: the host retries every rejection unconditionally, and being
53
+ // unconditional is correct there because a job retry is cheap. A boot refusal is not (issue #316).
54
+ if (isDeterminateFetchFailure(err)) throw configError(`could not resolve the forgejo bot identity from ${url}: ${fetchFailureReason(err)}`);
55
+ throw transientError(`could not resolve the forgejo bot identity from ${url}: ${fetchFailureReason(err)}`, err);
51
56
  }
52
57
  if (res.status === 403 || res.status === 401) {
53
58
  // The likely cause, named. A repo-scoped Forgejo token cannot carry `read:user`, so this is the
@@ -58,13 +63,27 @@ export async function resolveForgejoSelfId({ apiUrl, token, botId = null, fetchF
58
63
  }
59
64
  if (!res.ok) {
60
65
  // The status only. A Forgejo error body can echo the request, and the request carried the token.
66
+ // 401 and 403 were answered above, where they name the scope fix; what is left splits on the
67
+ // shared rule, so a self-hosted instance answering 502 mid-restart no longer reads as a
68
+ // misconfiguration the operator has to go and find.
69
+ if (isTransientStatus(res.status, responseHeaderReader(res))) {
70
+ throw transientError(`could not resolve the forgejo bot identity: GET /user returned ${res.status}`);
71
+ }
61
72
  throw configError(`could not resolve the forgejo bot identity: GET /user returned ${res.status}`);
62
73
  }
63
74
  let body;
64
75
  try {
65
76
  body = await res.json();
66
77
  } catch (err) {
67
- throw configError(`could not resolve the forgejo bot identity: unparseable JSON from GET /user (${err?.message ?? "unknown"})`);
78
+ // A body that will not parse is ambiguous in the expensive direction, so the CONTENT TYPE
79
+ // decides it. A truncated JSON body is transient. An HTML page is not: it is a Cloudflare
80
+ // Access or OIDC portal answering instead of the forge, and no amount of restarting gets past
81
+ // one (issue #316).
82
+ const contentType = responseHeaderReader(res);
83
+ if (contentType("content-type") !== undefined && !isJsonContentType(contentType)) {
84
+ throw configError(`could not resolve the forgejo bot identity: GET /user answered ${String(contentType("content-type"))} instead of JSON, so something other than the Forgejo API replied`);
85
+ }
86
+ throw transientError(`could not resolve the forgejo bot identity: unparseable JSON from GET /user (${err?.message ?? "unknown"})`, err);
68
87
  }
69
88
  const id = body?.id;
70
89
  if (!Number.isInteger(id)) {
package/src/get-token.mjs CHANGED
@@ -28,6 +28,7 @@ import { Octokit as RealOctokit } from "@octokit/rest";
28
28
  import { configError } from "./config.mjs";
29
29
  import { resolveSelfId } from "./identity.mjs";
30
30
  import { InfraRetry } from "./processor.mjs";
31
+ import { isDeterminateFetchFailure, isDeterminateFsCode, isTransientStatus, octokitHeaderReader } from "./transient.mjs";
31
32
 
32
33
  /**
33
34
  * Build the auth surface for `cfg = { source, patVar?, appId?, installationId?, privateKeyPath? }`.
@@ -138,14 +139,39 @@ function requireToken(raw, what) {
138
139
  return token;
139
140
  }
140
141
 
141
- /** Run `gh auth token` (array args, no shell) and return the trimmed token, or throw configError. */
142
+ /**
143
+ * Run `gh auth token` (array args, no shell) and return the trimmed token.
144
+ *
145
+ * THIS RUNS PER JOB, not only at boot: `makeGitHubAuth`'s gh arm mints once to resolve the identity and
146
+ * then again on every delivery, so a fault here is a fault on the job path.
147
+ *
148
+ * `promisify(execFile)` overloads `error.code`, and issue #316 is what that overloading cost. A SPAWN
149
+ * failure sets it to a string errno; a non-zero EXIT sets it to the integer status. The old comment
150
+ * called both deterministic and tagged both, so `EAGAIN` and `EMFILE` under fd pressure -- the host being
151
+ * momentarily busy -- were reported to the issue author as "is the gh CLI installed and logged in?" and
152
+ * never retried.
153
+ *
154
+ * So: a spawn errno splits on the same allow-list every other absence question uses, and `ENOENT` (no
155
+ * binary on PATH) stays determinate. A NON-ZERO EXIT stays determinate too, and that is a judgment call
156
+ * rather than a derivation. `gh auth token` exits 1 for a logged-out host far more often than for a
157
+ * locked keyring or a token refresh that could not reach the network, and the only thing that separates
158
+ * them is a table of gh's own stderr strings, which is exactly the hand-maintained table this project
159
+ * keeps deleting. The residual is named here rather than papered over.
160
+ */
142
161
  async function runGhAuthToken(execFileAsync) {
143
162
  let stdout;
144
163
  try {
145
164
  ({ stdout } = await execFileAsync("gh", ["auth", "token"]));
146
165
  } catch (error) {
147
- // ENOENT (binary absent) or a non-zero exit (logged out / broken) -- both deterministic.
148
166
  const detail = error?.code ?? error?.message ?? "unknown";
167
+ // A child killed by a SIGNAL sets `code` to null and `signal` to the name. An OOM-killed `gh` is
168
+ // the host being under pressure, not a deployment that is wrong.
169
+ if (error?.signal) {
170
+ throw new InfraRetry(`\`gh auth token\` was killed by ${error.signal}; the host is under pressure`);
171
+ }
172
+ if (typeof error?.code === "string" && !isDeterminateFsCode(error.code)) {
173
+ throw new InfraRetry(`\`gh auth token\` could not be run (${detail}); the host is momentarily unable to spawn it`);
174
+ }
149
175
  throw configError(`\`gh auth token\` failed (${detail}); is the gh CLI installed and logged in?`);
150
176
  }
151
177
  // Logged-out gh can exit 0 with empty stdout; an empty token is the money hole, so refuse it.
@@ -177,30 +203,53 @@ function repoNameOf(repo) {
177
203
 
178
204
  /**
179
205
  * Map an octokit/auth-app rejection to the retry-vs-config distinction the queue depends on.
180
- * Retryable (InfraRetry): 429, 403 + Retry-After (secondary rate limit), 5xx, and status-less
181
- * network faults (ENOTFOUND/ECONNRESET/ETIMEDOUT). Deterministic (configError): 401 bad
182
- * credentials, 404 unknown installation, and any other 4xx. Collapsing all 4xx to config would
183
- * turn a transient rate-limit into a permanent failure, so the classes are kept disjoint.
206
+ *
207
+ * Determinate (configError): 401 bad credentials and 404 unknown installation, both of which refuse
208
+ * identically until an operator changes something. Everything the shared rule calls transient is an
209
+ * `InfraRetry`; every other 4xx is determinate.
210
+ *
211
+ * TWO CORRECTIONS TO WHAT THIS COMMENT USED TO CLAIM (issue #316), because the old version described a
212
+ * function that did not exist.
213
+ *
214
+ * It said 403-with-retry-after was the rate-limit case. That is the SECONDARY limit. GitHub answers the
215
+ * PRIMARY limit -- the one a busy deployment actually hits, and which clears within the hour -- with 403,
216
+ * `x-ratelimit-remaining: 0` and no `retry-after`, so it fell through to the catch-all and became a
217
+ * permanent refusal. 408 and 425 went the same way, and so did an unparseable status, which
218
+ * `@octokit/request-error` reports as the NUMBER 0.
219
+ *
220
+ * It also said status-less network faults reach the bottom branch. They do not:
221
+ * `@octokit/request`'s fetch wrapper turns every network rejection into `RequestError(message, 500)`, so
222
+ * that branch is unreachable through octokit, and the 5xx arm has been doing the work all along.
223
+ *
224
+ * TWO CONSEQUENCES OF THAT SAME FACT, and they run in opposite directions.
225
+ *
226
+ * A TRUST OR REDIRECT FAULT ARRIVES WEARING A 500. The wrapper keeps the original rejection as `cause`,
227
+ * so the chain is intact, but the status says 500 and 500 is transient. A private CA in front of a GHES
228
+ * instance, or a TLS-inspecting corporate proxy re-signing `api.github.com`, would otherwise be retried
229
+ * forever instead of naming `NODE_EXTRA_CA_CERTS`. So the determinate check runs FIRST.
230
+ *
231
+ * AND AN ABSENT STATUS MEANS THIS NEVER REACHED THE NETWORK. Octokit always sets one, so a status-less
232
+ * rejection came from local code: `@octokit/auth-app` signing the JWT with a key that is not PKCS//8, the
233
+ * commonest App setup mistake there is. That is determinate, and it used to be retried and then reported
234
+ * as an infrastructure failure with the real reason nowhere an operator would look.
184
235
  */
185
236
  function classifyAppMintError(error) {
186
237
  const status = typeof error?.status === "number" ? error.status : undefined;
187
- const retryAfter = error?.response?.headers?.["retry-after"];
238
+ const detail = error?.message ?? "unknown";
188
239
 
240
+ if (isDeterminateFetchFailure(error)) return configError(`app token mint refused: ${detail}`);
189
241
  if (status === 401) return configError("app token mint refused: bad app credentials (401)");
190
242
  if (status === 404) return configError("app token mint refused: unknown installation (404)");
191
- if (status === 429) return new InfraRetry("app token mint: rate limited (429)");
192
- if (status === 403 && retryAfter !== undefined) {
193
- return new InfraRetry("app token mint: secondary rate limit (403 + retry-after)");
194
- }
195
- if (status !== undefined && status >= 500) {
196
- return new InfraRetry(`app token mint: upstream error (${status})`);
243
+ if (status === undefined) {
244
+ // Local, and therefore the operator's. `code` is worth carrying: a bad PEM surfaces as an OpenSSL
245
+ // DECODER error whose code says more than its message does.
246
+ const code = error?.code ? `${error.code}: ` : "";
247
+ return configError(`app token mint failed before any request was made (${code}${detail})`);
197
248
  }
198
- if (status !== undefined) {
199
- return configError(`app token mint failed (${status}): ${error?.message ?? "unknown"}`);
249
+ if (isTransientStatus(status, octokitHeaderReader(error), detail)) {
250
+ return new InfraRetry(`app token mint: transient upstream refusal (${status})`);
200
251
  }
201
- // No HTTP status -> network-level fault. Retry per INT-RUNNER-EXIT-CODE-PROTOCOL.
202
- const detail = error?.code ? `${error.code}: ` : "";
203
- return new InfraRetry(`app token mint: network fault: ${detail}${error?.message ?? "unknown"}`);
252
+ return configError(`app token mint failed (${status}): ${detail}`);
204
253
  }
205
254
 
206
255
  /**
package/src/git-dirty.mjs CHANGED
@@ -1,14 +1,22 @@
1
1
  import { execFileSync } from "node:child_process";
2
+ import { GIT_READ_FLAGS } from "./git-hardening.mjs";
2
3
 
3
4
  /**
4
5
  * Report a folder's git working-tree state: `true` = dirty, `false` = clean, `null` = not a usable
5
6
  * git repository. Reads the WORKING TREE via `git status --porcelain` — distinct from
6
7
  * flow-gate.mjs's object-store read, which reads committed content at a pinned SHA; the two are kept
7
8
  * separate on purpose. `exec` is injectable for tests.
9
+ *
10
+ * HARDENED, and this is the site where it is not hypothetical. `git status` REFRESHES THE INDEX, so it
11
+ * invokes `core.fsmonitor` -- an operator-supplied command -- where `rev-parse` does not. The folder this
12
+ * reads is the same path a local job mounts at /workspace with write access, so the agent can write
13
+ * `.git/config` and the next `pi-dispatch run` on that folder executes it ON THE HOST, outside any
14
+ * container. This file was invisible to the sweep that hardened its six siblings, because that census
15
+ * grepped for the flags that were PRESENT and this one had none.
8
16
  */
9
17
  export function gitDirty(folder, { exec = execFileSync } = {}) {
10
18
  try {
11
- const out = exec("git", ["-C", folder, "status", "--porcelain"], { encoding: "utf8" });
19
+ const out = exec("git", [...GIT_READ_FLAGS, "-C", folder, "status", "--porcelain"], { encoding: "utf8" });
12
20
  return out.trim().length > 0;
13
21
  } catch {
14
22
  return null;
@@ -0,0 +1,33 @@
1
+ /**
2
+ * The `-c` pairs that stop a hostile repository config executing code on the worker HOST, outside any
3
+ * container. `materialize.mjs` states the intent in one line and it is the whole of it: no hooks, no
4
+ * external filters, no pager.
5
+ *
6
+ * A LEAF that imports nothing, on `container-spec.mjs`'s precedent and for its reason. `materialize.mjs`
7
+ * is the file three "keep in sync" comments already named as canonical, but it imports `flow-gate.mjs`
8
+ * and `flow-gate.mjs` needs these flags too, so canonical-by-comment could not become
9
+ * canonical-by-import without a cycle. The array moves here and the comments become imports.
10
+ *
11
+ * Seven sites carried this by hand and one of them -- `prepare-local.mjs` -- was missing
12
+ * `core.fsmonitor`. Nothing was exploitable: its only command is `git rev-parse HEAD`, which does not
13
+ * refresh the index, so the fsmonitor hook is never invoked, and the `.pi/` materialisation beside it
14
+ * uses materialize's own hardened git. That is exactly the shape of a drift that survives review --
15
+ * harmless in the copy that has it wrong, load-bearing in the six that have it right -- and it is why
16
+ * this file exists rather than an eighth careful copy.
17
+ *
18
+ * `core.fsmonitor` is the one worth naming: it makes git run an operator-supplied command on any
19
+ * index-refreshing read, and a local-folder job's agent can write `.git/config` inside `/workspace`, so
20
+ * the value is attacker-controlled by construction on exactly the path that reads it.
21
+ */
22
+ export const GIT_SAFE_CONFIG = Object.freeze(["-c", "core.hooksPath=/dev/null", "-c", "core.fsmonitor=false"]);
23
+
24
+ /**
25
+ * The whole prefix for a READ that must run no repository-supplied code. `-C <dir>` follows at the call
26
+ * site, because two callers pass a `-C` and one (doctor's probe) supplies its own directory handling.
27
+ *
28
+ * Two constants rather than one: `prepare-github.mjs` INTERLEAVES its clone-specific locks
29
+ * (`protocol.ext.allow`, an empty `credential.helper`) between the fsmonitor pair and `--no-pager`, so a
30
+ * single flat prefix would not compose there and would force that file's argv to be reordered for the
31
+ * convenience of this one.
32
+ */
33
+ export const GIT_READ_FLAGS = Object.freeze([...GIT_SAFE_CONFIG, "--no-pager"]);
@@ -235,16 +235,28 @@ export async function runGithubAppSetup(argv = [], deps = {}) {
235
235
  fs.writeFileSync(envPath, "");
236
236
  say(`note: no .env existed here — created one for the lines you approved\n`);
237
237
  }
238
- updateEnvFile(envPath, "GITHUB_AUTH_SOURCE", "app", { fs, overwrite: true });
239
- updateEnvFile(envPath, "GITHUB_APP_ID", String(app.id), { fs, overwrite: true });
240
- updateEnvFile(envPath, "GITHUB_APP_PRIVATE_KEY_PATH", pemPath, { fs, overwrite: true });
241
- summary.push(["auth source", "GITHUB_AUTH_SOURCE=app (+ app id and key path) written to .env"]);
242
- if (offerSecret) {
243
- // setEnvKeyIfEmpty, NOT overwrite: an operator's existing webhook secret is what their
244
- // already-configured hooks sign with — replacing it would invalidate working deliveries.
245
- const { changed } = updateEnvFile(envPath, "WEBHOOK_SECRET", app.webhook_secret, { fs });
246
- say(changed ? "✓ WEBHOOK_SECRET set in .env (value not shown)\n" : "✓ WEBHOOK_SECRET already set in .env — kept (your configured hooks keep verifying)\n");
247
- summary.push(["WEBHOOK_SECRET", changed ? "set from the App's minted secret (value not shown)" : "already set — kept, so existing deliveries stay valid"]);
238
+ // WRAPPED, because `updateEnvFile` can now refuse rather than write: a `.env` owned by another
239
+ // account or another group, and a value no loader of that file can read back. Unwrapped, the
240
+ // refusal would leave a half-done setup -- the PEM already on disk at 0600, possibly an empty
241
+ // `.env` just created above -- and a stack trace instead of the summary this command exists to
242
+ // print. The App itself is already created on GitHub by this point, so saying exactly what landed
243
+ // is the whole remaining value.
244
+ try {
245
+ updateEnvFile(envPath, "GITHUB_AUTH_SOURCE", "app", { fs, overwrite: true });
246
+ updateEnvFile(envPath, "GITHUB_APP_ID", String(app.id), { fs, overwrite: true });
247
+ updateEnvFile(envPath, "GITHUB_APP_PRIVATE_KEY_PATH", pemPath, { fs, overwrite: true });
248
+ summary.push(["auth source", "GITHUB_AUTH_SOURCE=app (+ app id and key path) written to .env"]);
249
+ if (offerSecret) {
250
+ // setEnvKeyIfEmpty, NOT overwrite: an operator's existing webhook secret is what their
251
+ // already-configured hooks sign with — replacing it would invalidate working deliveries.
252
+ const { changed } = updateEnvFile(envPath, "WEBHOOK_SECRET", app.webhook_secret, { fs });
253
+ say(changed ? "✓ WEBHOOK_SECRET set in .env (value not shown)\n" : "✓ WEBHOOK_SECRET already set in .env — kept (your configured hooks keep verifying)\n");
254
+ summary.push(["WEBHOOK_SECRET", changed ? "set from the App's minted secret (value not shown)" : "already set — kept, so existing deliveries stay valid"]);
255
+ }
256
+ } catch (err) {
257
+ say(`error: could not write the App's lines into ${envPath}: ${err?.message}\n`);
258
+ say(` → the App exists on GitHub and its key is at ${pemPath}. Add GITHUB_AUTH_SOURCE=app, GITHUB_APP_ID=${app.id} and GITHUB_APP_PRIVATE_KEY_PATH=${pemPath} to that file by hand\n`);
259
+ summary.push(["auth source", `NOT written: ${err?.message}`]);
248
260
  }
249
261
  say(`✓ credentials written\n`);
250
262
 
@@ -320,7 +332,12 @@ export async function runGithubAppSetup(argv = [], deps = {}) {
320
332
  // so this one IS printed in full.
321
333
  say(`\nsetup would write to ${envPath}:\n GITHUB_APP_INSTALLATION_ID=${chosen.id}${chosen.account?.login ? ` (account: ${chosen.account.login})` : ""}\n`);
322
334
  if (await consent(prompt, "write it? [y/N] ")) {
323
- updateEnvFile(envPath, "GITHUB_APP_INSTALLATION_ID", String(chosen.id), { fs, overwrite: true });
335
+ try {
336
+ updateEnvFile(envPath, "GITHUB_APP_INSTALLATION_ID", String(chosen.id), { fs, overwrite: true });
337
+ } catch (err) {
338
+ say(`error: could not write GITHUB_APP_INSTALLATION_ID into ${envPath}: ${err?.message}\n`);
339
+ say(` → add GITHUB_APP_INSTALLATION_ID=${chosen.id} to that file by hand\n`);
340
+ }
324
341
  say("✓ GITHUB_APP_INSTALLATION_ID written\n");
325
342
  summary.push(["installation id", `GITHUB_APP_INSTALLATION_ID=${chosen.id} written to .env`]);
326
343
  } else {
@@ -346,7 +363,7 @@ function printSummary(say, summary, { noWebhook }) {
346
363
  say(" Polling delivery is a follow-up; until it lands, forge triggers for this App will not fire on their own.\n");
347
364
  } else {
348
365
  say(" start the receiver so deliveries have somewhere to land: `pi-dispatch-receiver`\n");
349
- say(" (or the deploy/docker-compose.yml receiver profile), reachable at the --webhook-url you gave GitHub.\n");
366
+ say(" (or, where the folder holds deploy/docker-compose.yml, its receiver profile), reachable at the --webhook-url you gave GitHub.\n");
350
367
  }
351
368
  say(" `pi-dispatch doctor` re-checks the whole deployment, including these credentials.\n");
352
369
  }
@@ -3,7 +3,10 @@
3
3
  *
4
4
  * ISOLATION BOUNDARY (read before touching the delimiter below):
5
5
  * The string this returns is written to /job/prompt.md and handed to session.prompt() as the USER
6
- * prompt — never a system prompt, never appendSystemPrompt (see image/runner/run-job.mjs:21,37,93).
6
+ * prompt: never a system prompt, never appendSystemPrompt. See image/runner/run-job.mjs:105 (read from
7
+ * /job/prompt.md) and :367 (`session.prompt(prompt)`), and image/runner/src/loader.mjs:366, whose
8
+ * appendSystemPromptOverride carries only the guardrails, the outbox protocol and the personas (why it
9
+ * overrides rather than discovers: :219-224).
7
10
  * That placement IS the control: issue/PR text is data because it enters as a user turn, after the
8
11
  * persona and the baked HARD_RULES system prompt, which the model treats as authoritative
9
12
  * (CONST-ISSUE-TEXT-IS-DATA). The `## Triggering …` heading and the code fence around the payload are
@@ -26,6 +26,7 @@
26
26
  import { configError } from "./config.mjs";
27
27
  import { InfraRetry } from "./processor.mjs";
28
28
  import { fetchFailureReason } from "./gitlab-identity.mjs";
29
+ import { isDeterminateFetchFailure } from "./transient.mjs";
29
30
 
30
31
  const API_PREFIX = "/api/v4";
31
32
 
@@ -39,6 +40,15 @@ export function makeGitLabHost({ apiUrl = "https://gitlab.com", fetchFn = fetch
39
40
  try {
40
41
  res = await fetchFn(`${root}${path}`, { headers: { "PRIVATE-TOKEN": token }, redirect: "error" });
41
42
  } catch (err) {
43
+ // The one condition where the identity modules and this one now agree, because the issue's
44
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
45
+ // that always redirects are the operator's, and retrying them twice before failing tells
46
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
47
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
48
+ // the delivery and posts publicly that the deployment is misconfigured.
49
+ if (isDeterminateFetchFailure(err)) {
50
+ throw configError(`gitlab-host: GET ${path} failed (${fetchFailureReason(err)})`);
51
+ }
42
52
  throw new InfraRetry(`gitlab-host: GET ${path} failed (${fetchFailureReason(err)})`);
43
53
  }
44
54
  if (res.status === 404 && notFound !== undefined) return notFound;
@@ -122,6 +132,15 @@ export function makeGitLabHost({ apiUrl = "https://gitlab.com", fetchFn = fetch
122
132
  redirect: "error",
123
133
  });
124
134
  } catch (err) {
135
+ // The one condition where the identity modules and this one now agree, because the issue's
136
+ // acceptance asks for exactly that (#316): a TLS trust failure, a protocol mismatch and a URL
137
+ // that always redirects are the operator's, and retrying them twice before failing tells
138
+ // nobody anything. Everything else here stays unconditionally retryable, which is right for a
139
+ // per-job call behind the queue: a wrong retry costs one more attempt, a wrong refusal costs
140
+ // the delivery and posts publicly that the deployment is misconfigured.
141
+ if (isDeterminateFetchFailure(err)) {
142
+ throw configError(`gitlab-host: POST ${path} failed (${fetchFailureReason(err)})`);
143
+ }
125
144
  throw new InfraRetry(`gitlab-host: POST ${path} failed (${fetchFailureReason(err)})`);
126
145
  }
127
146
  if (!res.ok) {
@@ -16,6 +16,7 @@
16
16
  */
17
17
 
18
18
  import { configError } from "./config.mjs";
19
+ import { isDeterminateFetchFailure, isJsonContentType, isTransientStatus, responseHeaderReader, transientError } from "./transient.mjs";
19
20
 
20
21
  /** Resolve the acting identity's integer user id from `GET /user`. `fetchFn` is injected for tests. */
21
22
  export async function resolveGitLabSelfId({ apiUrl, token, fetchFn = fetch }) {
@@ -27,17 +28,33 @@ export async function resolveGitLabSelfId({ apiUrl, token, fetchFn = fetch }) {
27
28
  try {
28
29
  res = await fetchFn(url, { headers: { "PRIVATE-TOKEN": token }, redirect: "error" });
29
30
  } catch (err) {
30
- throw configError(`gitlab identity: GET /user failed (${fetchFailureReason(err)})`);
31
+ // Unreachable is not misconfigured. The adjacent gitlab-host.mjs has classified this as
32
+ // InfraRetry since it was written; this file disagreed, and the disagreement decided whether
33
+ // the receiver ever came back from a forge restart (issue #316).
34
+ if (isDeterminateFetchFailure(err)) throw configError(`gitlab identity: GET /user failed (${fetchFailureReason(err)})`);
35
+ throw transientError(`gitlab identity: GET /user failed (${fetchFailureReason(err)})`, err);
31
36
  }
32
37
  if (!res.ok) {
33
38
  // The status alone, never the body: an error body can echo the token back.
39
+ if (isTransientStatus(res.status, responseHeaderReader(res))) {
40
+ throw transientError(`gitlab identity: GET /user returned ${res.status}`);
41
+ }
34
42
  throw configError(`gitlab identity: GET /user returned ${res.status}`);
35
43
  }
36
44
  let body;
37
45
  try {
38
46
  body = await res.json();
39
47
  } catch (err) {
40
- throw configError(`gitlab identity: GET /user returned unparseable JSON (${err?.message ?? "unknown"})`);
48
+ // A truncated body or a proxy interstitial, not a deployment an operator can fix.
49
+ // A body that will not parse is ambiguous in the expensive direction, so the CONTENT TYPE
50
+ // decides it. A truncated JSON body is transient. An HTML page is not: it is a Cloudflare
51
+ // Access or OIDC portal answering instead of the forge, and no amount of restarting gets past
52
+ // one (issue #316).
53
+ const contentType = responseHeaderReader(res);
54
+ if (contentType("content-type") !== undefined && !isJsonContentType(contentType)) {
55
+ throw configError(`gitlab identity: GET /user answered ${String(contentType("content-type"))} instead of JSON, so something other than the GitLab API replied`);
56
+ }
57
+ throw transientError(`gitlab identity: GET /user returned unparseable JSON (${err?.message ?? "unknown"})`, err);
41
58
  }
42
59
  if (!Number.isInteger(body?.id)) {
43
60
  throw configError("gitlab identity: GET /user returned no integer id");
@@ -135,6 +135,17 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
135
135
  // supersede lease `wait:key:<dedupId>` -- and only after checking the lease is still ours, which is
136
136
  // the same ownership check this key gets for free by being named after its only writer.
137
137
  await bounded(redis.pexpire(key, ttlMs), timeoutMs);
138
+ // RE-CHECKED HERE, AND ONLY HERE (issue #302). `close`'s drain is bounded at one `timeoutMs` while
139
+ // this function can spend one per command, so a write that outlived the drain would SADD the name
140
+ // back AFTER `close`'s SREM had removed it: the ghost the `inFlight` drain exists to prevent,
141
+ // arriving by a different door. Measured: `hset@0, pexpire@74, del@104, srem@105, sadd@145`.
142
+ //
143
+ // NOT before the PEXPIRE, which is the tempting symmetry and is wrong. A close landing between the
144
+ // HSET and the PEXPIRE would then leave `host:h:<name>` with NO EXPIRY AT ALL, where today it dies in
145
+ // ninety seconds -- invisible, because every reader walks `host:live` and the SREM emptied it, so a
146
+ // keyspace leak rather than a ghost peer, and the reader's own prune does not reach it. The SADD is
147
+ // the index write `close` undoes; the PEXPIRE is the row's own fuse and is always worth attempting.
148
+ if (closed) return;
138
149
  await bounded(redis.sadd(HOST_SET, name), timeoutMs);
139
150
  };
140
151
 
@@ -149,12 +160,22 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
149
160
  try {
150
161
  inFlight = write({ ...facts, name, beatAt: now() });
151
162
  await inFlight;
152
- if (!reachable) {
163
+ // GATED ON `closed`, IN BOTH ARMS, and that is the whole of issue #302. `close` drains with ONE
164
+ // `bounded(inFlight, timeoutMs)` while `write` spends a fresh `timeoutMs` on each of its commands,
165
+ // so a write can outlive the drain -- and the continuation here then speaks through the boot's own
166
+ // `log` closure, stamped with a host that has stopped. Measured against a stalled Valkey:
167
+ // `host_registry_unreachable` at +76ms and `host_registry_restored` at +108ms after `close()` had
168
+ // RESOLVED. The restored arm is not the afterthought it looks like: a drain that gave up can see the
169
+ // write SUCCEED late just as easily as fail, and #302 named only the catch.
170
+ //
171
+ // This is `makeWatchCloser`'s `reloadLog` posture reached from the other side: work already in
172
+ // flight cannot be recalled, so what a close gates is its VOICE.
173
+ if (!reachable && !closed) {
153
174
  reachable = true;
154
175
  log("host_registry_restored", { host: name });
155
176
  }
156
177
  } catch (err) {
157
- if (reachable) {
178
+ if (reachable && !closed) {
158
179
  reachable = false;
159
180
  log("host_registry_unreachable", { host: name, reason: err?.message });
160
181
  }
@@ -194,9 +215,9 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
194
215
 
195
216
  /**
196
217
  * Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
197
- * `setTimeout` -- and `.unref()`'d so it can never hold the process open, which is the posture the
198
- * three `fs.watch` watchers already take. `stop` is registered as an extraCloser beside the runtime
199
- * queue, so a clean shutdown clears it before `process.exit`.
218
+ * `setTimeout` -- and `.unref()`'d so it can never hold the process open. `close` is registered as an
219
+ * extraCloser beside the runtime queue, so a clean shutdown clears it before `process.exit`; the three
220
+ * `fs.watch` watchers take the same two-part posture since issue #295, unref'd AND closed.
200
221
  */
201
222
  async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
202
223
  if (closed || timer) return; // a second start would leak the first interval
@@ -218,6 +239,12 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
218
239
  timer = null;
219
240
  // Drain before deleting, bounded like everything else here: an unbounded wait on a beat that is
220
241
  // itself hung would be the shutdown hang this module's timeout exists to prevent.
242
+ //
243
+ // So the drain is BEST-EFFORT, and it deliberately stays at one `timeoutMs` (issue #302): `write`
244
+ // can spend one per command, so this can return while a write is still running, and widening the
245
+ // wait to match would triple a shutdown bound whose entire point is to be short. What makes that
246
+ // safe is the `closed` FLAG rather than the wait -- it gates the SADD inside `write` and both log
247
+ // arms in `beat`, so a write that outlives this drain can neither re-index the row nor say a word.
221
248
  await bounded(inFlight ?? Promise.resolve(), timeoutMs).catch(() => {});
222
249
  try {
223
250
  await bounded(redis.del(hostKey(name)), timeoutMs);
package/src/identity.mjs CHANGED
@@ -2,8 +2,10 @@
2
2
  * Resolve the acting GitHub identity's numeric `id` -- the value that appears in webhook
3
3
  * `sender.id`. The receiver's bot-loop guard compares an incoming `sender.id` against this to
4
4
  * refuse events the harness itself authored (an unbounded paid recursion otherwise). If the id
5
- * cannot be resolved the guard cannot arm, so this fails CLOSED: any failure throws a tagged
6
- * config error and the process must not boot.
5
+ * cannot be resolved the guard cannot arm, so this fails CLOSED: any failure throws and the
6
+ * process must not boot. WHICH failure it throws is what decides whether the process comes back
7
+ * (issue #316): a determinate fault is tagged and stays stopped, a transient one is untagged and
8
+ * is restarted.
7
9
  *
8
10
  * The `octokit` client is INJECTED, already authenticated by the caller (user token for pat/gh,
9
11
  * app-JWT for app). This module never constructs Octokit -- keeping it a pure, testable leaf, per
@@ -11,6 +13,7 @@
11
13
  */
12
14
 
13
15
  import { configError } from "./config.mjs";
16
+ import { isDeterminateFetchFailure, isTransientStatus, octokitHeaderReader, transientError } from "./transient.mjs";
14
17
 
15
18
  /**
16
19
  * Resolve the acting identity's numeric user id from `auth = { source, octokit }`.
@@ -20,8 +23,24 @@ import { configError } from "./config.mjs";
20
23
  * up in `sender.id` is `slug[bot]`, resolved via `GET /users/{username}`. The App id is a
21
24
  * different number and would never match `sender.id`, so the two-step is required.
22
25
  *
23
- * Returns an integer id. Throws `configError` on unknown/missing source, missing octokit, any
24
- * octokit rejection, or a non-integer id.
26
+ * Returns an integer id. Throws `configError` on unknown/missing source, missing octokit, a determinate
27
+ * octokit refusal, or a non-integer id.
28
+ *
29
+ * A TRANSIENT octokit rejection throws UNTAGGED instead (issue #316). The tag is not decoration here: it
30
+ * is the difference between a service that comes back and one that does not. `receiver/src/cli.mjs` maps
31
+ * a tagged throw to `EXIT_POLICY` (2), which `RestartPreventExitStatus=2` and nssm's `AppExit 2 Exit`
32
+ * deliberately leave stopped; and on the worker, whose `cli.mjs` never sees this because `start.mjs`
33
+ * catches it best-effort, the tag decides whether that forge stays credential-less for the lifetime of
34
+ * the process, after which every job of that kind is publicly told the deployment is misconfigured. A
35
+ * `GET /user` that timed out is none of those things.
36
+ *
37
+ * TWO THINGS THE STATUS ALONE CANNOT TELL YOU, both of which have to be answered before it is consulted.
38
+ * `@octokit/request`'s fetch wrapper turns EVERY network rejection into `RequestError(message, 500)` and
39
+ * keeps the original as `cause`, so a private CA or a redirect arrives wearing a transient status and the
40
+ * determinate check has to run first. And an ABSENT status means the failure never reached the HTTP layer
41
+ * at all, because octokit always sets one: it came from local code, which in practice means signing the
42
+ * App JWT with a key that is not PKCS//8 -- the commonest App setup mistake there is, and one that no
43
+ * amount of restarting fixes.
25
44
  */
26
45
  export async function resolveSelfId(auth) {
27
46
  const source = auth?.source;
@@ -47,6 +66,12 @@ export async function resolveSelfId(auth) {
47
66
  id = user.id;
48
67
  }
49
68
  } catch (error) {
69
+ // Order is load-bearing; see the docblock. Trust and redirect faults are wearing a 500, and a
70
+ // status-less rejection never reached the network.
71
+ const status = typeof error?.status === "number" ? error.status : undefined;
72
+ if (!isDeterminateFetchFailure(error) && status !== undefined && isTransientStatus(status, octokitHeaderReader(error), error?.message)) {
73
+ throw transientError(`resolveSelfId: could not reach GitHub to resolve self identity: ${error.message}`, error);
74
+ }
50
75
  throw configError(`resolveSelfId: could not resolve self identity: ${error.message}`);
51
76
  }
52
77