@edgehero/pi-dispatch 1.10.3 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.example +303 -150
  2. package/README.md +52 -0
  3. package/deploy/com.pi-dispatch.worker.plist +10 -4
  4. package/deploy/docker-compose.yml +49 -16
  5. package/deploy/egress-proxy.conf +32 -2
  6. package/deploy/nssm-install.cmd +12 -6
  7. package/deploy/pi-dispatch-egress-out.network +10 -0
  8. package/deploy/pi-dispatch-egress-proxy.container +50 -0
  9. package/deploy/pi-dispatch-netns-keeper.container +80 -0
  10. package/deploy/pi-dispatch-netns-keeper.network +18 -0
  11. package/deploy/pi-dispatch-valkey.container +51 -0
  12. package/deploy/pi-dispatch-valkey.network +16 -0
  13. package/deploy/receiver.service +6 -0
  14. package/deploy/worker-env-wrapper.cmd +12 -1
  15. package/deploy/worker-env-wrapper.sh +63 -37
  16. package/deploy/worker.service +18 -8
  17. package/package.json +15 -5
  18. package/src/azure-host.mjs +19 -0
  19. package/src/azure-identity.mjs +18 -2
  20. package/src/backend-conformance.mjs +71 -18
  21. package/src/backend-local.mjs +637 -21
  22. package/src/backend-podman.mjs +1168 -0
  23. package/src/backend-registry.mjs +86 -3
  24. package/src/backends.mjs +489 -37
  25. package/src/branch.mjs +7 -2
  26. package/src/cancel-cli.mjs +174 -0
  27. package/src/cancel-state.mjs +125 -0
  28. package/src/cli.mjs +188 -90
  29. package/src/config.mjs +503 -43
  30. package/src/connection.mjs +374 -8
  31. package/src/container-spec.mjs +102 -7
  32. package/src/daemon-facts.mjs +167 -0
  33. package/src/deployment-venue.mjs +158 -0
  34. package/src/docker-run.mjs +146 -15
  35. package/src/doctor.mjs +4756 -394
  36. package/src/egress-conf-copy.mjs +166 -0
  37. package/src/egress-proxy-state.mjs +151 -0
  38. package/src/egress.mjs +456 -25
  39. package/src/entry.mjs +27 -0
  40. package/src/env-allowlist.mjs +245 -40
  41. package/src/env-file.mjs +1869 -33
  42. package/src/exit-code.mjs +15 -0
  43. package/src/flow-gate.mjs +5 -3
  44. package/src/forgejo-host.mjs +19 -0
  45. package/src/forgejo-identity.mjs +21 -2
  46. package/src/get-token.mjs +67 -18
  47. package/src/git-dirty.mjs +9 -1
  48. package/src/git-hardening.mjs +33 -0
  49. package/src/github-app-setup.mjs +29 -12
  50. package/src/github-prompt.mjs +4 -1
  51. package/src/gitlab-host.mjs +19 -0
  52. package/src/gitlab-identity.mjs +19 -2
  53. package/src/host-pi.mjs +19 -3
  54. package/src/host-registry.mjs +29 -2
  55. package/src/identity.mjs +29 -4
  56. package/src/image-preflight.mjs +46 -11
  57. package/src/image-ref.mjs +21 -0
  58. package/src/index.mjs +363 -13
  59. package/src/init.mjs +197 -38
  60. package/src/job-user.mjs +252 -0
  61. package/src/json-duplicates.mjs +204 -0
  62. package/src/live-probes.mjs +1020 -0
  63. package/src/materialize.mjs +4 -11
  64. package/src/netns-keeper.mjs +264 -0
  65. package/src/on-failure.mjs +119 -0
  66. package/src/outbox.mjs +7 -0
  67. package/src/packages.mjs +2 -2
  68. package/src/podman-stack.mjs +1304 -0
  69. package/src/prepare-github.mjs +6 -6
  70. package/src/prepare-local.mjs +51 -17
  71. package/src/prepare.mjs +27 -6
  72. package/src/pricing.mjs +9 -5
  73. package/src/processor.mjs +506 -26
  74. package/src/provider-key.mjs +66 -0
  75. package/src/provider-steering.mjs +185 -0
  76. package/src/queue.mjs +35 -8
  77. package/src/redact.mjs +84 -0
  78. package/src/reserved-env.mjs +7 -3
  79. package/src/retention-sweep.mjs +178 -0
  80. package/src/run-container.mjs +181 -14
  81. package/src/run-history.mjs +105 -16
  82. package/src/runtime-observations.mjs +1152 -0
  83. package/src/runtime-settings.mjs +13 -8
  84. package/src/sandbox-cli.mjs +100 -95
  85. package/src/sandbox-store.mjs +612 -45
  86. package/src/sandbox.mjs +1459 -37
  87. package/src/schedules.mjs +16 -3
  88. package/src/secret-profiles.mjs +2 -1
  89. package/src/secrets.mjs +24 -6
  90. package/src/service-env.mjs +247 -0
  91. package/src/service.mjs +618 -28
  92. package/src/session-store.mjs +678 -53
  93. package/src/start.mjs +1348 -326
  94. package/src/subscriptions.mjs +7 -3
  95. package/src/transient.mjs +240 -0
  96. package/src/triggers-file.mjs +71 -15
  97. package/src/triggers.mjs +179 -19
  98. package/src/up.mjs +1399 -85
  99. package/src/valkey-auth.mjs +529 -0
  100. package/src/valkey-endpoint.mjs +367 -0
  101. package/src/watch-closer.mjs +158 -0
package/src/processor.mjs CHANGED
@@ -1,10 +1,40 @@
1
- import { DEFAULT_BACKEND, DOCKER_NEVER_STARTED_EXITS } from "./backends.mjs";
1
+ import { DAEMON_APPLIES_BOUNDS, DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, DOCKER_NEVER_STARTED_EXITS, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_CONF_WIDENS_JOB, PODMAN_SERVICE_LOCAL, RUNTIME_ADDS_NO_MOUNTS } from "./backends.mjs";
2
+ import { resolveBackendName } from "./backend-registry.mjs";
2
3
  import { lstatSync } from "node:fs";
3
4
  import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
4
5
  import { configError } from "./config.mjs";
5
6
  import { scopeKeyPrefix } from "./scoped-limits.mjs";
6
7
  import { DEFAULT_SECRETS_PROFILE, secretsArmed } from "./secrets.mjs";
7
8
  import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
9
+ import { RUNNER_POLICY_REASONS } from "./run-history.mjs";
10
+ import { DEFAULT_EGRESS_PROXY } from "./egress.mjs";
11
+
12
+ /**
13
+ * The forge comment's reason for each observation a floor refusal missed (issues #278 and #345), keyed like
14
+ * `OBSERVATIONS` in `backends.mjs` and pinned to it. Fixed words only.
15
+ */
16
+ export const OBSERVATION_COMMENT = Object.freeze({
17
+ [DOCKER_ENDPOINT_LOCAL]: "the docker CLI is not observed sending containers to a daemon on this host, so the job's credentials could cross a network the deployment does not own",
18
+ [DAEMON_APPLIES_BOUNDS]: "the container runtime is not observed applying a container's pid and memory bounds",
19
+ [RUNTIME_ADDS_NO_MOUNTS]: "the container runtime is not observed adding no mounts of its own to a job container",
20
+ // Issue #354: the podman venue's three, in the same fixed register. No path, no controller list and no service URL:
21
+ // those are the evidence, which goes to the operator's log only.
22
+ // Cause-neutral (issue #453): the observation misses both when a controller is not delegated and when no systemd user
23
+ // manager runs for the account (measured: podman info then still lists every controller). Which one is the evidence's
24
+ // to say, in the operator's log.
25
+ [PODMAN_BOUNDS_DELEGATED]: "the worker's rootless Podman is not observed applying a container's pid, memory and cpu bounds",
26
+ [PODMAN_ADDS_NO_MOUNTS]: "the worker's rootless Podman is not observed adding no mounts of its own to a job container",
27
+ [PODMAN_SERVICE_LOCAL]: "the podman CLI is not observed running containers on this host rather than through a remote service, so the job's credentials could cross a network the deployment does not own",
28
+ });
29
+
30
+ /**
31
+ * The forge comment's reason when a floor refusal names no observation this build has words for (issue #354): a
32
+ * venue other than `local` that named none, or named one outside `OBSERVATIONS`. Venue-neutral on purpose. The
33
+ * fallback used to be the docker endpoint's sentence, which is right only for `local`, whose preflight is the one
34
+ * that could ever refuse without naming what it missed; told to a job on another runtime it blames a CLI that job
35
+ * never touched. Not a key of `OBSERVATION_COMMENT`, which is pinned to the closed list one-for-one.
36
+ */
37
+ export const OBSERVATION_COMMENT_UNNAMED = "the venue this job runs on did not confirm a guarantee the floor requires";
8
38
 
9
39
  /**
10
40
  * The job orchestration. Deliberately a pure-ish function over INJECTED side-effecting deps, so
@@ -13,6 +43,13 @@ import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
13
43
  * The order is the contract, and every step before `runContainer` must be free of provider spend:
14
44
  *
15
45
  * 0. refuse a job image this host does not have -- INT-CONTAINER-RUNTIME-CONTRACT
46
+ * 0a. refuse a job no non-root uid can run on this daemon, or whose image cannot run as the worker's uid
47
+ * (issue #341), and retry one whose job user could not be decided -- DES-JOB-USER-INFERRED-READ-BACK-ON-REQUEST
48
+ * 0b. REFUSE a deployment with no usable credential for the job's provider, which costs nothing to
49
+ * ask and would otherwise be discovered with the budget already reserved -- CONST-BUDGET-BEFORE-TOKENS
50
+ * (gates 0 and 0b are not the first two: a one-shot already spent, a skewed wait, an unblessed
51
+ * backend and a backend floor this host is not observed to meet (#278, #345) are refused above them, and
52
+ * this ladder has never listed those)
16
53
  * 1. REFUSE an armed `run.resume` with no session store to persist into (the one fail-CLOSED case)
17
54
  * -- REQ-RESUMABLE-SESSION
18
55
  * 2. mint a scoped token (GitHub jobs, and local jobs opted in via `github: true`)
@@ -33,6 +70,62 @@ import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
33
70
  * retries per `attempts`. The caller (the BullMQ processor) turns the thrown/returned distinction
34
71
  * into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
35
72
  */
73
+
74
+ // The post-spend terminal comments (issue #288). Every FREE refusal above the container already comments;
75
+ // these are the paths where money was spent and the run still ended without the agent's own status step,
76
+ // which used to tell the issue nothing (REQ-JOB-STATUS-COMMENTS' acceptance -- "exactly one completion or
77
+ // failure comment" -- held only below the spend line). Keyed by the reason token so a new abort
78
+ // classification adds a ROW here, never a re-plumb; sentences are FIXED and path-free, because for a
79
+ // local job the adapter logs the full text into a persistent service log (the "this folder" discipline
80
+ // at the scope-cap refusal). The completed path stays silent on purpose: exit 0 is where the AGENT'S own
81
+ // status comment lives (the prompt contract instructs it, including for "I cannot fix this"), and exit 2
82
+ // by construction means the agent was cut off before that step.
83
+ // EXPORTED only so a test can hold every RUNNER_POLICY_REASONS member to a row here (issue #437 review).
84
+ export const TERMINAL_COMMENTS = {
85
+ "worker-abort": "Stopped: the worker ended this run before it finished (the 30-minute job limit, or a worker shutdown). Partial work may exist. Not retried.",
86
+ "operator-cancel": "Stopped: the operator cancelled this run. Partial work may exist. Not retried.",
87
+ "runner-policy": "Stopped: the run ended inside the container before finishing (a turn or token budget, or an in-container configuration refusal). Partial work may exist. Not retried.",
88
+ // Issue #437. Names the cause but never the provider's own message, which may echo a key fragment. "Or
89
+ // access" because a 403 is as often a key that works but may not use this model or route as a bad key.
90
+ "provider-auth-refused": "Stopped: the AI provider refused this worker's credentials or access (an authentication or permission error). The operator needs to check the provider key and what it is allowed to use. Not retried.",
91
+ };
92
+
93
+ // Issue #341: the forge comments for a `job-user-unmappable` refusal, keyed by cause. Shorter than the operator
94
+ // texts in job-user.mjs on purpose: a comment's reader may be an issue author, who can act on none of it.
95
+ const JOB_USER_COMMENTS = Object.freeze({
96
+ default: "Refused: the worker host's container runtime cannot give this job a non-root user that can read and write its own files. Not run.",
97
+ "runtime-unreadable": "Refused: the worker host's container runtime answered in a form the worker cannot read, so which user this job may run as is unknown. Not run.",
98
+ });
99
+
100
+ /**
101
+ * The runtime a venue's own preflights talk to, named in a retry's words (issue #354). The message is the job_failed log
102
+ * line and BullMQ's failedReason, so `local`'s stays byte-identical ("docker unavailable, ..."), the native podman venue
103
+ * names podman (an operator reading "docker unavailable" on a host with no docker would chase the wrong daemon), and a
104
+ * venue this build has no runtime word for gets the neutral phrase rather than a guess. `Object.hasOwn`, so a venue
105
+ * named like a prototype key is not a runtime.
106
+ */
107
+ const VENUE_RUNTIME = Object.freeze({ [DEFAULT_BACKEND]: "docker", [PODMAN_BACKEND]: "podman" });
108
+
109
+ function runtimeUnavailable(venue, gate) {
110
+ return Object.hasOwn(VENUE_RUNTIME, venue) ? `${VENUE_RUNTIME[venue]} unavailable, ${gate} could not run` : `the container runtime is unavailable, ${gate} could not run`;
111
+ }
112
+
113
+ /**
114
+ * The remedy an egress-proxy refusal's comment names. The compose profile starts the proxy on the DOCKER daemon, which a
115
+ * job on the native podman venue cannot reach: rootless podman's `--internal` network reaches nothing on the host
116
+ * (measured, issue #354), so its proxy must run under the SAME rootless podman on a named bridge network, which is what
117
+ * docs/podman.md sets up. Every other venue names `pi-dispatch up` from the deployment folder, and ONLY that (issue #480,
118
+ * PR #488's review): it starts the proxy in any folder init made, and it knows the folder. A compose line printed from
119
+ * here could not: a folder /dispatch setup laid out needs its project and override, and without them `--profile egress
120
+ * up -d` also starts compose's unprofiled valkey under project `deploy`, a second Valkey on a fresh volume or a clash on
121
+ * its port. A proxy PI_EGRESS_PROXY names is the operator's own, which `up` never starts, so that one is theirs to start.
122
+ */
123
+ function egressProxyFix(venue, proxy = DEFAULT_EGRESS_PROXY) {
124
+ if (venue === PODMAN_BACKEND) return "Start it under the worker account's own rootless podman, on a named bridge network (docs/podman.md)";
125
+ if (proxy !== DEFAULT_EGRESS_PROXY) return `PI_EGRESS_PROXY names your own proxy, which \`pi-dispatch up\` does not start: start ${proxy} yourself`;
126
+ return "Start it with `pi-dispatch up` from the deployment folder";
127
+ }
128
+
36
129
  export async function runJob(job, deps) {
37
130
  const {
38
131
  redis,
@@ -58,9 +151,25 @@ export async function runJob(job, deps) {
58
151
  // Issue #230. Admit-everything by default, like checkOnceSpent above and for its reason: an
59
152
  // unwired seam must not refuse, and the wiring is what turns the check on.
60
153
  checkWaitSkew = async () => ({ ok: true }),
154
+ // () => { ok } | { message }. Issue #310. Resolves this deployment's provider credential the way
155
+ // buildContainerEnv will, and answers whether it exists AT ALL, so an unconfigured provider refuses
156
+ // here rather than inside runContainer with the budget already reserved. Admit-everything by default,
157
+ // like the two above and for their reason. A PROBE, deliberately: it discards whatever it resolves and
158
+ // the real read happens where it always did, because threading a live credential through the processor
159
+ // would put it in scope for every log line and record between here and the container.
160
+ checkProviderCredential = () => ({ ok: true }),
61
161
  // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
62
162
  // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
63
163
  egressPreflight = async () => ({ ok: true }),
164
+ // Issue #278: a live read of something the backend's declaration holds only while observed (today, which
165
+ // docker endpoint the CLI resolves), judged against the deployment's floor. `{ ok }`, `{ refused, message }`
166
+ // or `{ unavailable }`. Defaults to ok, so a bare wiring gates nothing, like the two preflights above it.
167
+ observationPreflight = async () => ({ ok: true }),
168
+ // Issue #341: which uid this job's container runs as. `(job, { capabilities, observed }) =>` `{ user, home }`
169
+ // (`user` null = the image's own USER), `{ refused, cause }` or `{ unavailable, reason }`. The default runs
170
+ // every job as the image's user, exactly as before, so a wiring that omits it changes nothing. A non-refused answer
171
+ // may also carry `relabel: true` (issue #355), which reaches `runContainer` beside the user it was decided with.
172
+ jobUserPreflight = async () => ({ user: null, home: null }),
64
173
  // (session, { piVersion, context }) => { promoted, reason, bytes }. Promotes this job's transcript back into
65
174
  // the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
66
175
  // it behaves exactly as before -- no store, no promotion, no session in the record.
@@ -121,10 +230,15 @@ export async function runJob(job, deps) {
121
230
  mintToken,
122
231
  isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
123
232
  prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
124
- // runContainer({ job, token, prepared, secrets, name, signal }) => { code, aborted, turns, tokens, session, usage, context }.
233
+ // runContainer({ job, token, prepared, secrets, name, signal, user, home }) => { code, aborted, abortReason, turns, tokens, session, usage, context, exitReason }.
234
+ // `exitReason` (issue #437) is parseExitReason's closed-set label, read only inside the exit-2 branch.
235
+ // `user`/`home` are the job-user gate's answer (issue #341), null for the image's own USER.
125
236
  // `secrets` is the resolved map from the gate above: values, already fetched, host-side. It MUST honour
126
237
  // `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
127
- // true at entry (the timeout can fire during a slow prepare). The wiring injects name + signal.
238
+ // true at entry (the timeout can fire during a slow prepare). The wiring injects name + signal, and
239
+ // on an aborted result it also maps `signal.reason` onto `abortReason` (issue #287) so the
240
+ // classification below can tell an operator's cancel from the kill timer's; a bare wiring that
241
+ // never sets it classifies every abort as worker-abort, exactly as before.
128
242
  runContainer,
129
243
  cleanup, // (dirs) => void
130
244
  comment, // (job, text) => void (issue status; no-op for local jobs)
@@ -149,6 +263,18 @@ export async function runJob(job, deps) {
149
263
  let prepared = null;
150
264
  let reserved = false;
151
265
  let scopedReserved = false;
266
+ // Set once `runContainer` has RESOLVED, which is the only moment a container is known to have run. The
267
+ // config classifier in the catch refunds both ledgers, and its whole justification is that nothing was
268
+ // spent; without a fact to test, that is a claim about where config throws happen to live today rather
269
+ // than a property of the code. A config-tagged throw raised after a paid run would otherwise refund a slot
270
+ // the container really spent AND tell the operator publicly that nothing was.
271
+ //
272
+ // AFTER the await and not before, deliberately: `runContainer` resolves the credential and assembles the
273
+ // env before it spawns anything, so its config throws (an unconfigured provider, an unknown forge kind)
274
+ // happen with no container started at all -- and those are exactly what the classifier exists to refund.
275
+ // A spawn fault that fails between is an InfraRetry carrying `container-never-started`, refunded by the
276
+ // arm below, which has a discriminator of its own.
277
+ let containerRan = false;
152
278
 
153
279
  try {
154
280
  // The one-shot pre-spend check (issue #231), FIRST on the ladder: one file read, cheaper than
@@ -213,6 +339,114 @@ export async function runJob(job, deps) {
213
339
  return { outcome: "policy", reason: "backend-unblessed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
214
340
  }
215
341
 
342
+ // Issue #278. PI_BACKEND_FLOOR can ask for a guarantee a backend declares but EARNS only while something
343
+ // about this host is observed -- `local`'s credentialTransit, which holds only while the docker CLI sends
344
+ // containers to a daemon on this host. The read is repeated here, per job, because the CLI's context can
345
+ // change after boot and every later job's provider key and forge token would follow it. FREE and
346
+ // pre-spend: a refusal that will recur on every job must not cost each one a slot. Determinate is a
347
+ // RETURN (CONST-RETRY-INFRA-ONLY); a CLI that could not be asked for a transient reason is a throw,
348
+ // pre-reserve, so the refund is a no-op.
349
+ //
350
+ // AHEAD of the image and egress preflights, not beside them, because both of those talk to the daemon
351
+ // this gate is about: on a redirected CLI an unreachable daemon made the image inspect throw a retry, and
352
+ // a reachable one without the image refused as `job-image-missing` with a comment blaming the image --
353
+ // the right refusal lost behind the wrong one. Issue #345 adds `isolation` (the daemon observed applying a
354
+ // container's bounds) and `mountSet` (the runtime observed adding no mounts), read here from the same
355
+ // `docker info` the job user is decided from, cached per endpoint once it answers, so this read now contacts the
356
+ // daemon (after the endpoint check, which still refuses without it).
357
+ // One refusal, reached from two places (the venue's own answer below, or the job user decided after the image
358
+ // probe). Fixed text per cause class: the cause and anything the runtime said go to the operator's log, never a
359
+ // forge comment.
360
+ const refuseJobUserUnmappable = async (cause) => {
361
+ await comment(job, JOB_USER_COMMENTS[cause] ?? JOB_USER_COMMENTS.default);
362
+ log("refused_job_user_unmappable", { cause: cause ?? null });
363
+ return { outcome: "policy", reason: "job-user-unmappable", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
364
+ };
365
+ const observed = await observationPreflight(job);
366
+ if (observed?.refused) {
367
+ // Fixed text per observation: the endpoint (an internal host name or address) and the evidence go to the
368
+ // operator's log, never to a forge comment. "Not observed" rather than "not on this host", because a CLI that
369
+ // could not be asked for a determinate reason (no docker on PATH, a context that does not exist) refuses too.
370
+ // A refusal naming nothing keeps the endpoint's words on `local` only (#278); elsewhere it is the neutral
371
+ // sentence, and so is a name this build has no words for, on any venue (issue #354).
372
+ const onLocal = resolveBackendName(job, blessedBackends[0]) === DEFAULT_BACKEND;
373
+ const missed = Array.isArray(observed.observations) && observed.observations.length > 0 ? observed.observations : [onLocal ? DOCKER_ENDPOINT_LOCAL : null];
374
+ const why = missed.map((o) => (Object.hasOwn(OBSERVATION_COMMENT, o ?? "") ? OBSERVATION_COMMENT[o] : OBSERVATION_COMMENT_UNNAMED));
375
+ await comment(job, `Refused: this deployment's PI_BACKEND_FLOOR requires a guarantee this host is not observed to provide right now (${[...new Set(why)].join("; ")}). Not run.`);
376
+ log("refused_backend_floor_unobserved", { message: observed.message });
377
+ return {
378
+ outcome: "policy",
379
+ reason: "backend-floor-unobserved",
380
+ exitCode: null,
381
+ turns: null,
382
+ tokens: null,
383
+ provider: job.provider ?? null,
384
+ model: job.model ?? null,
385
+ budgetReserved: false,
386
+ };
387
+ }
388
+ if (observed?.unavailable) {
389
+ // The local venue's words stay what they were (they are its log line and BullMQ's failedReason); another venue
390
+ // is not docker's to name.
391
+ const localVenue = resolveBackendName(job, blessedBackends[0]) === DEFAULT_BACKEND;
392
+ // A host FILE that could not be read for a moment (issue #428) is named as that, never as a runtime outage.
393
+ const why = typeof observed.message === "string" && observed.message !== "" ? `a host file an observation the floor needs could not be read just now (${observed.message})` : localVenue ? "docker CLI or daemon unavailable, an observation the floor needs could not run" : "the container runtime or its CLI is unavailable, an observation the floor needs could not run";
394
+ throw new InfraRetry(why, { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
395
+ }
396
+
397
+ // A venue that already knows no uid can run a job there says so HERE, before the image preflight asks that same
398
+ // runtime (issue #354): behind it, a runtime that is not there is retried as unavailable and one that answers
399
+ // without the image is refused as `job-image-missing`, both the wrong fix. Only an answer no image can change
400
+ // comes this way; the per-image half (`anyUid`) stays below, after the probe that reads it.
401
+ if (observed?.jobUserRefused?.refused === "job-user-unmappable") return refuseJobUserUnmappable(observed.jobUserRefused.cause);
402
+ // Issue #428: the podman venue's account sets a containers.conf key that widens what every job reaches (the host's
403
+ // loopback services, the account's groups) and no argv takes back. A VENUE refusal, not a floor miss: with egress
404
+ // off no declared property covers what a job reaches on the host, so a floor would never ask on the deployments at
405
+ // risk. Determinate (the account's own files), so a RETURN (CONST-RETRY-INFRA-ONLY), here for the identity's reason:
406
+ // ahead of the image preflight and every spend. The file and key go to the operator's log; the comment is fixed,
407
+ // and says the configuration widens a job only when a key was FOUND: a chain that could not be read whole is
408
+ // refused for not being known, which is a different sentence. A read that failed for a moment (`transient`) is
409
+ // infrastructure, so it throws and is retried, pre-reserve, exactly like an unanswered observation.
410
+ if (observed?.podmanConfRefused?.transient) {
411
+ // `evidence` names the file (`<path> could not be read (<errno>)`), so the retry says which one: a containers.conf,
412
+ // or since issue #450 the /proc entry of the account's running rootless network. Issue #448: `rootful` is the local
413
+ // venue's, a containers.conf or unit file of rootful Podman's service on this host.
414
+ const what = observed.podmanConfRefused.rootful ? "rootful Podman's containers.conf or podman.service" : "the podman venue's containers.conf or running rootless network";
415
+ throw new InfraRetry(`${what} could not be read just now, so whether it widens a job is not known (${observed.podmanConfRefused.evidence ?? "no file named"})`, { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
416
+ }
417
+ if (observed?.podmanConfRefused?.retry) {
418
+ // Gate round 1 of PR #473: rootful Podman's service running since before its containers.conf changed (or a change
419
+ // time ahead of the clock) heals by itself, once the service restarts or idles out or the clock passes, so it is
420
+ // never a final `policy` outcome, pre-reserve. Gate round 2: a HOLD, not a retry (`PodmanRestartHold`), which the
421
+ // processor moves back to the delayed set without spending an attempt, for up to `PODMAN_RESTART_HOLD_MAX_MS`.
422
+ throw new PodmanRestartHold(`rootful Podman's service may still hold a containers.conf older than the files, so this job waits for it to restart (${observed.podmanConfRefused.evidence ?? "no file named"})`, { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
423
+ }
424
+ if (observed?.podmanConfRefused) {
425
+ // Issue #450: `live` is the account's RUNNING rootless network, not a file, so it has its own two sentences: one
426
+ // still carrying an option a removed key gave it (the fix is a restart, not a configuration change), and one
427
+ // whose process could not be read.
428
+ const { key, live, rootful } = observed.podmanConfRefused;
429
+ // Issue #448: `rootful` is the LOCAL venue on rootful Podman's Docker API service, with its own sentences: a key
430
+ // that reaches every local job, and a file it cannot read or does not decode. (A service older than its
431
+ // configuration is a retry, above, since gate round 1 of PR #473.)
432
+ await comment(
433
+ job,
434
+ rootful
435
+ ? key
436
+ ? "Refused: the worker host's Podman configuration adds to every job's container what this venue does not allow (variables, the Podman service's groups, looser limits and filters, or the programs that run it), so the operator must change it before local jobs run. Not run."
437
+ : "Refused: the worker host's Podman configuration could not be read in full, or is written in a form the worker does not decode, so whether it adds to a job's container what this venue does not allow is not known, and the operator must fix that before local jobs run. Not run."
438
+ : live
439
+ ? key
440
+ ? "Refused: the worker host's running Podman network still lets a job's container reach the host's own services, with an option from a configuration since changed, so the operator must restart that network before podman jobs run. Not run."
441
+ : "Refused: the worker host's running Podman network could not be read, so whether it lets a job's container reach more than this venue allows is not known, and the operator must fix that before podman jobs run. Not run."
442
+ : key
443
+ ? "Refused: the worker host's Podman configuration lets a job's container reach more than this venue allows (the host's own services, the worker account's groups, or looser limits and filters than the worker sets), so the operator must change it before podman jobs run. Not run."
444
+ : "Refused: the worker host's Podman configuration could not be read in full, or is written in a form the worker does not decode, so whether it lets a job's container reach more than this venue allows is not known, and the operator must fix that before podman jobs run. Not run.",
445
+ );
446
+ log("refused_podman_conf_widens_job", { key: key ?? null, ...(live ? { live: true } : {}), ...(rootful ? { rootful: true } : {}), message: observed.podmanConfRefused.message });
447
+ return { outcome: "policy", reason: PODMAN_CONF_WIDENS_JOB, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
448
+ }
449
+
216
450
  // The job image must exist on THIS host before anything else happens. Free, determinate and
217
451
  // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
218
452
  // image refuses without minting a credential it will not use, cloning a repo it will not read, or
@@ -287,14 +521,90 @@ export async function runJob(job, deps) {
287
521
  log("refused_image_commands_unsupported", { image: img.commandUnsupported, declared: img.declared });
288
522
  return { outcome: "policy", reason: "job-image-commands-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
289
523
  }
524
+ if (img.excludeToolsUnsupported) {
525
+ // The image is present and does not declare exclude-tools support (issue #291), so its runner
526
+ // predates run.excludeTools: it reads no PI_EXCLUDE_TOOLS, and the job would run with every
527
+ // tool the trigger says to remove -- a "read-only" trigger with a working editor and shell,
528
+ // recording a clean exit. That is a PERMISSION quietly not enforced, the silent fail-open this
529
+ // repo brands the worst outcome available, which is exactly why the host refuses before spend
530
+ // rather than letting the container fail open.
531
+ //
532
+ // Determinate, so a refusal rather than a retry, and pre-spend, because no version of this
533
+ // gets better by running. Like the command branch above, the message names the FIX rather
534
+ // than the label that noticed it.
535
+ await comment(
536
+ job,
537
+ `Refused: the job image "${img.excludeToolsUnsupported}" does not declare exclude-tools support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its runner would ignore \`run.excludeTools\` and run this trigger with every tool it says to remove. Rebuild the image from a version that has this feature. Not run.`,
538
+ );
539
+ log("refused_image_exclude_tools_unsupported", { image: img.excludeToolsUnsupported, declared: img.declared });
540
+ return { outcome: "policy", reason: "job-image-exclude-tools-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
541
+ }
290
542
  if (img.unavailable) {
291
543
  // docker itself did not answer -- transient infra, NOT a determinate refusal. THROWN so BullMQ
292
544
  // retries (CONST-RETRY-INFRA-ONLY). `container-never-started` is literally true here, and it reuses
293
545
  // the refund path below: a no-op pre-reserve, and still honest if this gate ever moves.
294
546
  // provider/model attribute even this pre-container death; no usage -- nothing ran to emit one.
295
- throw new InfraRetry("docker unavailable, image preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
547
+ throw new InfraRetry(runtimeUnavailable(resolveBackendName(job, blessedBackends[0]), "image preflight"), { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
296
548
  }
297
549
 
550
+ // Issue #341: WHO runs the container. On a daemon that enforces bind-mount ownership the job must run as the
551
+ // uid that owns its job dir and mounts, or it cannot read its own inputs; where no uid works (a rootless
552
+ // daemon, userns-remap, a root worker) nothing runs. After the image probe because the answer needs its
553
+ // `anyUid` capability, and still FREE: one cached `docker info` per endpoint, no mint, no clone, no reserve.
554
+ const jobUser = await jobUserPreflight(job, { capabilities: img.capabilities ?? [], observed });
555
+ if (jobUser?.refused === "job-user-unmappable") return refuseJobUserUnmappable(jobUser.cause);
556
+ if (jobUser?.refused === "job-image-any-uid-unsupported") {
557
+ // The image ref is operator config, the same PII class as the refusals above.
558
+ await comment(
559
+ job,
560
+ `Refused: the job image "${img.image}" does not declare \`anyUid\` (\`dev.pi-dispatch.capabilities\`), so it cannot run as this worker's own uid, which this host's container runtime requires. Rebuild the image from a version that has this feature. Not run.`,
561
+ );
562
+ log("refused_job_image_any_uid_unsupported", { image: img.image });
563
+ return { outcome: "policy", reason: "job-image-any-uid-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
564
+ }
565
+ if (jobUser?.unavailable) {
566
+ // The reason is a fixed token (`daemon-unreachable`, `timeout`, `endpoint-unresolved`...), never CLI text.
567
+ log("job_user_unavailable", { reason: jobUser.reason ?? null });
568
+ throw new InfraRetry("the job user could not be decided", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
569
+ }
570
+
571
+ // Is there a credential to run this job with at all? FREE, determinate and I/O-light: a pure function
572
+ // of the job's provider, the worker env, and (only when the env has no key) one small readFileSync of
573
+ // pi's auth.json. So it goes here, with the other free gates, which is further than issue #310 asked
574
+ // for: ahead of the image probe, the mint, the clone, the token-cap read and both reserves. That is
575
+ // CONST-BUDGET-BEFORE-TOKENS in its own words, "every gate that costs nothing runs before every gate
576
+ // that costs something".
577
+ //
578
+ // AFTER the image probe and not before, which is the one ordering choice here that costs something:
579
+ // a `docker inspect` runs before a `readFileSync`. This file says twice that a missing image blocks
580
+ // EVERY job of EVERY kind on this host, so it is the fault an operator must fix first either way, and
581
+ // a missing credential is in exactly that class -- on a host with both, the reported reason should be
582
+ // the one that was already the rule. Everything the gate is actually FOR is still below it: the egress
583
+ // probe, the secret resolvers, the mint, the clone, the token-cap read and both reserves.
584
+ //
585
+ // It used to be discovered inside runContainer, where buildContainerEnv resolves the credential for
586
+ // real -- AFTER both reserves -- and the throw fell through this function's catch to a bare rethrow.
587
+ // The catch now classifies that too (below), so this gate is the cheap path and that is the backstop;
588
+ // neither alone is enough, because the backstop cannot un-mint a token or un-clone a repository.
589
+ //
590
+ // The refusal names no path. `credentialFromPiAuth`'s messages carry auth.json's location, `comment`
591
+ // posts publicly on the issue, and the reason an operator needs is the same either way: their
592
+ // deployment has no usable provider credential and `doctor` will say exactly which variable.
593
+ const credential = await checkProviderCredential(job);
594
+ if (!credential.ok) {
595
+ // The PROVIDER is named and the message is not. The provider is operator-authored config, already
596
+ // on the record and the mirror, and named freely by the sibling refusals (`backend-unblessed` names
597
+ // the backend, `job-image-missing` names the image), so withholding it would tell the operator less
598
+ // than is safe. The message is withheld from BOTH surfaces: `credentialFromPiAuth`'s refusals carry
599
+ // `auth.json`'s absolute path, which is an OS account name, and the scoped-budget refusal below
600
+ // keeps a host path out of its own log line citing no-pii-in-logs. `doctor` prints the variable and
601
+ // the path, on the operator's terminal, which is where that belongs.
602
+ await comment(job, `Refused: this deployment has no usable credential for the "${job.provider}" provider, so no container was started and nothing was spent. Ask the operator to run \`pi-dispatch doctor\`. Not run.`);
603
+ log("refused_provider_unconfigured", { provider: job.provider ?? null });
604
+ return { outcome: "policy", reason: "provider-unconfigured", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
605
+ }
606
+
607
+
298
608
  // REQ-EGRESS-ALLOWLIST. The egress policy this deployment claims must be able to serve this job
299
609
  // BEFORE the job costs anything. It is one `docker inspect` when the policy is armed and ZERO spawns
300
610
  // when it is not, so a deployment without one pays nothing at all.
@@ -313,7 +623,7 @@ export async function runJob(job, deps) {
313
623
  if (egress.proxyMissing || egress.proxyStopped) {
314
624
  const proxy = egress.proxyMissing ?? egress.proxyStopped;
315
625
  const state = egress.proxyMissing ? "is not on this host" : "is not running";
316
- await comment(job, `Refused: this deployment runs jobs behind an egress policy and its allowlist proxy "${proxy}" ${state}, so the job could not reach the provider and would burn its budget slot proving it. Start it with \`docker compose -f deploy/docker-compose.yml --profile egress up -d\`, or set PI_EGRESS=0 to run without an egress policy. Not run.`);
626
+ await comment(job, `Refused: this deployment runs jobs behind an egress policy and its allowlist proxy "${proxy}" ${state}, so the job could not reach the provider and would burn its budget slot proving it. ${egressProxyFix(resolveBackendName(job, blessedBackends[0]), proxy)}, or set PI_EGRESS=0 to run without an egress policy. Not run.`);
317
627
  // The proxy's NAME is operator-authored deployment config, never payload -- the same PII class as
318
628
  // the image ref on the refusal above.
319
629
  log(egress.proxyMissing ? "refused_egress_proxy_missing" : "refused_egress_proxy_stopped", { proxy });
@@ -328,11 +638,27 @@ export async function runJob(job, deps) {
328
638
  budgetReserved: false, // refused before reserveBudget, so no job-count slot was consumed
329
639
  };
330
640
  }
641
+ if (egress.unavailable && egress.keeper) {
642
+ // Issue #458: the podman venue on Podman 4.x, proxy up, its rootless network keeper not holding. INFRA, not a
643
+ // refusal: the keeper is one `systemctl --user` away, and a retry after it holds runs the job. Pre-reserve, so
644
+ // nothing is refunded. Its OWN reason token (PR #463 round 2), so the run record, the failure hook and the
645
+ // terminal comment can name the keeper rather than a generic never-started container; the full sentence rides
646
+ // the error message and is logged whole here, where `job_failed` cuts it at 120 characters.
647
+ // Issue #476: a keeper whose only fault is its age is HELD for, not failed on. `makeProcessor` moves the job to
648
+ // the delayed set until the keeper is old enough, without an attempt, and names a crash loop if it keeps
649
+ // restarting; a path that does not know the hold still retries it, since it is an `InfraRetry`.
650
+ if (egress.young) throw new NetnsKeeperYoungHold(egress.keeper, { reason: NETNS_KEEPER_NOT_HOLDING, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false, young: egress.young, remedy: egress.remedy ?? null });
651
+ log("egress_keeper_not_holding", { proxy: egress.unavailable, reason: egress.keeper });
652
+ throw Object.assign(new InfraRetry(egress.keeper, { reason: NETNS_KEEPER_NOT_HOLDING, provider: job.provider ?? null, model: job.model ?? null }), { keeperProblem: egress.problem ?? null, keeperRemedy: egress.remedy ?? null });
653
+ }
331
654
  if (egress.unavailable) {
332
655
  // The daemon did not answer, so this is indeterminate rather than a refusal -- the same
333
656
  // determinate/indeterminate split the image preflight draws one gate up, and thrown for the same
334
657
  // reason. Pre-reserve, so the refund below is a no-op and still honest if this gate ever moves.
335
- throw new InfraRetry("docker unavailable, egress preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
658
+ // With a proxy STATE (issue #453, gate round 3): the daemon answered and the proxy is on its way somewhere
659
+ // (restarting, created, ...), so the words name the proxy and its state rather than blaming the runtime.
660
+ const said = typeof egress.state === "string" ? `egress proxy "${egress.unavailable}" is ${egress.state}, not running; the job is retried once, then failed` : runtimeUnavailable(resolveBackendName(job, blessedBackends[0]), "egress preflight");
661
+ throw new InfraRetry(said, { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
336
662
  }
337
663
 
338
664
  // REQ-RESUMABLE-SESSION's one fail-CLOSED case. Everything else in that feature fails OPEN and
@@ -421,8 +747,8 @@ export async function runJob(job, deps) {
421
747
  if (resolved.ambiguous) {
422
748
  // Two sources declared one profile name. Neither wins, deliberately: runtime-settings documents the
423
749
  // overlay's precedence as overlay > env, so inverting it here would leave two rules disagreeing about
424
- // what an overlay is, while honouring it would let a settings file in a world-writable default
425
- // directory redirect a profile the operator wrote in .env. This project already refuses ambiguity
750
+ // what an overlay is, while honouring it would let a settings file -- which PI_SETTINGS_FILE can put
751
+ // anywhere -- redirect a profile the operator wrote in .env. This project already refuses ambiguity
426
752
  // rather than resolving it (PI_EGRESS: "a typo must never leave you believing you have a policy you
427
753
  // do not"), and an operator who sees this fixes it in seconds.
428
754
  await comment(job, "Refused: this trigger set `run.secrets`, and the resolver profile it names is declared twice on this worker host, once in the environment and once in the settings overlay. Neither wins, on purpose: the job would otherwise run against whichever one happened to be picked. Remove one of the two. Not run.");
@@ -430,16 +756,22 @@ export async function runJob(job, deps) {
430
756
  return { outcome: "policy", reason: "secret-profile-ambiguous", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
431
757
  }
432
758
 
433
- if (resolved.reserved) {
434
- // A key the worker itself writes, and one parseTriggers could not have caught: the provider
435
- // credential's variable names depend on this job's resolved provider and on what this host has set,
759
+ if (resolved.reserved !== undefined) {
760
+ // A key the worker itself writes OR that pi reads for this job's provider, and one parseTriggers
761
+ // could not have caught: which variables the provider uses depends on this job's resolved provider,
436
762
  // and PI_FORWARD_ENV is an operator env list. Both are deployment state, so this is the same
437
763
  // load-time / pre-spend split run.resume makes against PI_SESSIONS_DIR.
438
764
  //
439
- // It matters most in the direction that is hardest to see: buildContainerEnv writes the provider
440
- // credential BEFORE this feature's values, so a trigger binding ANTHROPIC_API_KEY would silently
441
- // redirect which credential every job of that trigger spends.
442
- await comment(job, `Refused: this trigger's \`run.secrets\` binds \`${resolved.reserved}\`, and the worker sets that variable itself for every job. The container would receive the worker's value rather than this trigger's, and the trigger would look like it worked. Rename it in the triggers file. Not run.`);
765
+ // TWO failures, opposite in direction, which is why the message names neither (issue #309):
766
+ // - the worker WRITES the name: buildContainerEnv assigns the provider credential and
767
+ // PI_FORWARD_ENV before this feature's values, so the trigger's value replaces the operator's
768
+ // and every job of that trigger spends the trigger author's key;
769
+ // - the worker does NOT write the name but pi READS it first: the OAuth token variable, and from
770
+ // the 0.99.1 pin the bearer ANTHROPIC_AUTH_TOKEN (issue #509), are deliberately never written
771
+ // (apiKeyVariable skips both), so a trigger binding one lands beside the operator's key and
772
+ // outranks it in pi's own precedence.
773
+ // The old message asserted the first for both, which is exactly backwards for the second.
774
+ await comment(job, `Refused: this trigger's \`run.secrets\` binds \`${resolved.reserved}\`, which is a variable this deployment already uses for the job's own credentials. Whichever of the two values reached the container, one of them would be silently ignored. Rename it in the triggers file. Not run.`);
443
775
  // The variable NAME only. It is the operator's own choice of name, not payload, and naming it is what
444
776
  // makes the refusal actionable -- but the REFERENCE behind it never appears.
445
777
  log("refused_secret_name_reserved", { kind: job.kind ?? null, name: resolved.reserved });
@@ -495,7 +827,8 @@ export async function runJob(job, deps) {
495
827
  }
496
828
  }
497
829
 
498
- prepared = await prepareWorkspace(job, token, { piVersion }); // resolves SHA, clones, materialises .pi/, writes prompt
830
+ // `podmanStore` (issue #429) only where the venue's job user carried one: the podman store the container ran in.
831
+ prepared = await prepareWorkspace(job, token, { piVersion, jobUser: { user: jobUser?.user ?? null, home: jobUser?.home ?? null }, ...(typeof jobUser?.store === "string" ? { podmanStore: jobUser.store } : {}) }); // resolves SHA, clones, materialises .pi/, writes prompt
499
832
 
500
833
  // A determinate prepare refusal -- sha-gone (the default branch advanced past the resolved tip),
501
834
  // or a `pi-*` materialiser cap breach (the repo's .pi/ is too large to place in /job, issue #60)
@@ -568,6 +901,12 @@ export async function runJob(job, deps) {
568
901
  // never a refusal it did not issue.
569
902
  if (scopedReserved && scopedCaps) {
570
903
  await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
904
+ // CLEARED, so the ledger's state and the flag agree. Nothing between here and the return can
905
+ // throw today (`comment` is non-throwing by construction and `log` is a write), but the catch
906
+ // below now releases on a whole CLASS of error rather than one reason, and a second release
907
+ // against this same key would take it to -1: `releaseBudget` is a floorless DECR. The invariant
908
+ // belongs where the release is, not in the guard of every future reader.
909
+ scopedReserved = false;
571
910
  }
572
911
  const w = budget.blockedWindow;
573
912
  const win = budget.windows[w];
@@ -583,8 +922,10 @@ export async function runJob(job, deps) {
583
922
  return { outcome: "policy", reason: budget.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
584
923
  }
585
924
 
586
- const { code, aborted, turns, tokens, session, usage, context } = await runContainer({ job, token, prepared, secrets });
587
- log("container_exit", { exitCode: code, aborted });
925
+ // The user the gate above decided is the user that runs: one answer, never two call sites that agree.
926
+ const { code, aborted, abortReason, turns, tokens, session, usage, context, detached, exitReason } = await runContainer({ job, token, prepared, secrets, user: jobUser?.user ?? null, home: jobUser?.home ?? null, relabel: jobUser?.relabel === true });
927
+ containerRan = true;
928
+ log("container_exit", { exitCode: code, aborted, ...(detached === true ? { detached: true } : {}) });
588
929
 
589
930
  // Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
590
931
  // so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
@@ -597,13 +938,31 @@ export async function runJob(job, deps) {
597
938
  await recordSpend(redis, tokensSpent, { now }).catch((err) => log("token_spend_error", { reason: err?.message }));
598
939
  }
599
940
 
600
- // A WORKER-initiated stop (30-min timeout via cancelJob, or graceful-shutdown docker stop) kills
601
- // the container -> exit 143/137. That is our decision, not an infra fault: it is POLICY and must
602
- // NOT retry, or a wedged job re-runs into a second PR / double spend. Keyed on the abort FLAG,
603
- // not the code -- an unbidden 137 (kernel OOM) carries `aborted: false`, falls to the switch, and
604
- // stays infra-retryable.
941
+ // Issue #345: `docker run` exited with a never-started code, but THIS attempt's container was found by its cidfile,
942
+ // still there, and was stopped and removed (run-container.mjs, measured on Podman with its API service killed
943
+ // mid-job). It DID start, so this is never refunded as never-started: it keeps its slot and retries as infrastructure,
944
+ // BEFORE the exit-code switch, where the same code would read as a free never-started exit.
945
+ if (detached === true) {
946
+ throw new InfraRetry(`the container outlived its docker run, exit ${code}`, { reason: "container-detached", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
947
+ }
948
+
949
+ // A WORKER-initiated stop (30-min timeout via cancelJob, graceful-shutdown docker stop, or an
950
+ // operator's cancel, issue #287) kills the container -> exit 143/137. That is our decision, not an
951
+ // infra fault: it is POLICY and must NOT retry, or a wedged job re-runs into a second PR / double
952
+ // spend. Keyed on the abort FLAG, not the code -- an unbidden 137 (kernel OOM) carries
953
+ // `aborted: false`, falls to the switch, and stays infra-retryable.
954
+ // WHO aborted is an exact-match on `abortReason` (the wiring maps it off `signal.reason`), and the
955
+ // match is deliberately closed: "job-timeout-30m", "shutdown", undefined and any future garbage all
956
+ // classify as worker-abort, so a pin bump that changes what rides the signal can widen nothing.
957
+ // `abortReason` itself never reaches the record -- buildRecord copies named fields only.
605
958
  // exitCode/turns/tokens carry the container's own exit, turn count, and usage totals; budgetReserved true post-reserve.
606
- if (aborted) return { outcome: "policy", reason: "worker-abort", exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
959
+ if (aborted) {
960
+ const reason = abortReason === "operator-cancel" ? "operator-cancel" : "worker-abort";
961
+ // Awaited bare like every determinate refusal above: the adapter never throws by contract, and
962
+ // the one swallowed comment in this file (the catch's) justifies itself by its position.
963
+ await comment(job, TERMINAL_COMMENTS[reason]);
964
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
965
+ }
607
966
 
608
967
  switch (code) {
609
968
  case EXIT_COMPLETED: {
@@ -638,11 +997,34 @@ export async function runJob(job, deps) {
638
997
  chainRefused: chain.refused,
639
998
  };
640
999
  }
641
- case EXIT_POLICY:
1000
+ case EXIT_POLICY: {
642
1001
  // A policy exit still ran a paid container, so it carries the ledger like the completed
643
1002
  // branch does -- the spend is real whichever way the runner classified itself.
644
- return { outcome: "policy", reason: "runner-policy", exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
1003
+ // The comment is UNCONDITIONAL (issue #288 asked for "when the agent did not already comment
1004
+ // its own refusal", and the discriminator already exists at the exit-code boundary): an agent
1005
+ // that composed its own refusal exits 0 -- github-prompt.mjs instructs the status comment,
1006
+ // including for "I cannot fix this" -- so exit 2 means it was cut off before that step.
1007
+ // Residual: an agent that posted a status and THEN blew its turn budget yields one extra
1008
+ // comment, bounded at one.
1009
+ //
1010
+ // The runner's exit-line reason is read for ONE purpose (issue #437): a provider that refused
1011
+ // the credential needs an operator, not a wait, and "runner-policy" hid that behind budget
1012
+ // wording. It picks a label INSIDE this branch and nothing else, which is why reading it is
1013
+ // safe where reading it to classify would not be: `code` has already placed the job in the
1014
+ // not-retried class, so a forged or stale line can at worst swap one not-retried label for
1015
+ // another from the closed RUNNER_POLICY_REASONS set. The `code === 2` guard is redundant with
1016
+ // the case label today and is kept so the label cannot follow this line if it is ever moved.
1017
+ // Every other reason the runner gives still reads as runner-policy.
1018
+ const reason = RUNNER_POLICY_REASONS.has(exitReason) && code === 2 ? exitReason : "runner-policy";
1019
+ await comment(job, TERMINAL_COMMENTS[reason]);
1020
+ return { outcome: "policy", reason, exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
1021
+ }
645
1022
  case EXIT_INFRA:
1023
+ // NO comment on any infra throw, here or in the catch: an InfraRetry may be retried and
1024
+ // recover, and a flaky daemon must not post three comments for one recovery. Once-ness for
1025
+ // the whole infra class lives at the terminal seam -- start.mjs's failed listener, guarded on
1026
+ // BullMQ's own finishedOn -- which also catches the stall-kill and wait-gate paths this
1027
+ // function never sees (issue #288).
646
1028
  throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
647
1029
  default:
648
1030
  // THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
@@ -663,6 +1045,69 @@ export async function runJob(job, deps) {
663
1045
  throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
664
1046
  }
665
1047
  } catch (e) {
1048
+ // A CONFIG-tagged throw is a determinate policy refusal wearing an exception, and issue #310 is the
1049
+ // bill for treating it as neither. `CONST-RETRY-INFRA-ONLY` says a determinate refusal RETURNS and only
1050
+ // infrastructure THROWS; every other gate in this function obeys that, and this class did not, because
1051
+ // nothing here read the tag that `cli.mjs` and `doctor` both already read.
1052
+ //
1053
+ // What it actually cost, corrected against the issue text: it was NOT retried. `index.mjs` wraps every
1054
+ // non-InfraRetry throw in BullMQ's `UnrecoverableError`, so the job failed once. What it did cost is
1055
+ // the reserve, kept and never refunded, on a fault an operator has to fix by hand -- so every later
1056
+ // delivery took another slot out of the same daily cap and the same scope, and the record said
1057
+ // `outcome: "failed"` with a null reason, which the panel paints red beside real infrastructure faults
1058
+ // and insights buckets as a failure. A determinate refusal that reads as an outage is the diagnosis
1059
+ // this project exists to make legible.
1060
+ //
1061
+ // Refunding is right BECAUSE no container started: identical to `container-never-started`, and the
1062
+ // same both-or-neither pair, under the same `reserved`/`scopedReserved` guards so it still cannot
1063
+ // double-release. The gate above catches the provider case for free, before the mint and the clone;
1064
+ // this is the backstop for every other config throw that can still land here (an unknown forge kind in
1065
+ // `buildContainerEnv`, a prepare-time refusal), which would otherwise keep the same slot silently.
1066
+ // `!containerRan` is the discriminator, and it is what makes the refund and the sentence below TRUE
1067
+ // rather than merely true today. Every config-tagged throw site in the worker is pre-container, so this
1068
+ // changes nothing now; the day one is added after a paid run, that run keeps its slot and falls through
1069
+ // to the untagged path instead of being refunded and publicly declared free.
1070
+ if (e?.piDispatchConfig === true && !containerRan) {
1071
+ // GUARDED, and the flags are the record of what actually happened. `releaseBudget` is a loop of
1072
+ // DECRs over the active windows and can reject part-way (a read-only replica, a dropped
1073
+ // connection), which would otherwise replace this determinate refusal with a Redis message: the
1074
+ // operator would be told their queue is broken when their deployment is misconfigured, and the
1075
+ // escaping error is untagged so it is not retried either. A refund that did not land must not be
1076
+ // reported as one, so `budgetReserved` follows the ledger and not the intent.
1077
+ let refunded = true;
1078
+ try {
1079
+ if (reserved) {
1080
+ await releaseBudget(redis, { caps, now });
1081
+ reserved = false;
1082
+ }
1083
+ if (scopedReserved && scopedCaps) {
1084
+ await releaseBudget(redis, { caps: scopedCaps.caps, now, keyPrefix: scopeKeyPrefix(scopedCaps.scope) });
1085
+ scopedReserved = false;
1086
+ }
1087
+ } catch (releaseError) {
1088
+ refunded = false;
1089
+ log("budget_release_failed", { at: "config-refused", code: releaseError?.code ?? null });
1090
+ }
1091
+ // A FIXED sentence, and NO message in the log either. Two different reasons, both load-bearing:
1092
+ // `credentialFromPiAuth` puts `auth.json`'s location in its refusals and `prepare-local` puts the
1093
+ // operator's folder in its own, which `buildRecord` reduces to a basename precisely because a host
1094
+ // path carries an OS account name; and `branch.mjs`'s refusal interpolates a forge PAYLOAD field,
1095
+ // which no-pii-in-logs forbids anywhere. The scoped-budget refusal above keeps a host path out of
1096
+ // its log line for the first of those reasons, and this follows it. `doctor` is where the operator
1097
+ // reads the specific variable and the specific path, which is what both sentences point at.
1098
+ // SWALLOWED, and only here. Every other refusal in this function awaits `comment` bare, which is
1099
+ // right: they are on the happy path of a determinate decision, and a forge that cannot be told is
1100
+ // worth surfacing. This one is inside the catch, so a throw from `comment` would discard the
1101
+ // classification that has ALREADY released the budget, and the job would escape as whatever the
1102
+ // comment threw, with the ledger refunded and the record saying something else entirely. The
1103
+ // shipped adapter never throws; this makes that a property of the arm rather than of the wiring.
1104
+ await comment(job, "Refused: this deployment is misconfigured, so the job could not be started. Ask the operator to run `pi-dispatch doctor`. Not run.").catch(() => {});
1105
+ log("refused_config", { kind: job.kind ?? null, refunded });
1106
+ // budgetReserved reflects the LEDGER: false when the refund landed, true when it did not and the
1107
+ // slot is still out there.
1108
+ return { outcome: "policy", reason: "config-refused", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: !refunded }; // return => not retried
1109
+ }
1110
+
666
1111
  // A spawn fault (docker daemon down / binary missing) reserved a slot but never started a
667
1112
  // container, so nothing was spent -- give the slot back before the retry. Every other throw
668
1113
  // here (exit-1 infra, unknown exit) means the container ran and legitimately spent its slot,
@@ -684,6 +1129,14 @@ export async function runJob(job, deps) {
684
1129
  }
685
1130
  }
686
1131
 
1132
+ /**
1133
+ * The reason a job retried for the podman venue's rootless network keeper carries (issue #458, PR #463 round 2): a
1134
+ * fixed token, as every run-record reason is, and the key the terminal comment is chosen by.
1135
+ */
1136
+ export const NETNS_KEEPER_NOT_HOLDING = "netns-keeper-not-holding";
1137
+ /** Issue #476: a job held on a young keeper that kept restarting, or stayed young past the hold's bound. */
1138
+ export const NETNS_KEEPER_CRASH_LOOP = "netns-keeper-crash-loop";
1139
+
687
1140
  /** Thrown for the retryable (infra) class only. The BullMQ processor lets this propagate to retry. */
688
1141
  /**
689
1142
  * The one `session` object the run record carries, from the host's intent and the container's report
@@ -768,4 +1221,31 @@ export class InfraRetry extends Error {
768
1221
  }
769
1222
  }
770
1223
 
1224
+ /**
1225
+ * A local job held until rootful Podman's service restarts (issue #448, gate round 2 of PR #473). An `InfraRetry`, so
1226
+ * any path that does not know it still retries rather than failing the job for good; `makeProcessor` knows it, and
1227
+ * moves the job to the delayed set without spending an attempt until the hold has lasted `PODMAN_RESTART_HOLD_MAX_MS`.
1228
+ */
1229
+ export class PodmanRestartHold extends InfraRetry {
1230
+ constructor(message, options) {
1231
+ super(message, options);
1232
+ this.name = "PodmanRestartHold";
1233
+ this.holdUntilRestart = true;
1234
+ }
1235
+ }
1236
+
1237
+ /**
1238
+ * A job held on a rootless network keeper that is running on its own bridge but younger than the minimum age (issue
1239
+ * #476). An `InfraRetry`, so any path that does not know it still retries; `makeProcessor` knows it, and moves the job to
1240
+ * the delayed set for `young.waitMs` without spending an attempt, up to `NETNS_KEEPER_YOUNG_HOLD_MAX_MS`.
1241
+ */
1242
+ export class NetnsKeeperYoungHold extends InfraRetry {
1243
+ constructor(message, { young, remedy = null, ...options } = {}) {
1244
+ super(message, options);
1245
+ this.name = "NetnsKeeperYoungHold";
1246
+ this.keeperYoung = young;
1247
+ this.keeperRemedy = remedy;
1248
+ }
1249
+ }
1250
+
771
1251
  export { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY };