@edgehero/pi-dispatch 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.env.example +160 -0
  2. package/deploy/com.pi-dispatch.worker.plist +66 -0
  3. package/deploy/nssm-install.cmd +59 -0
  4. package/deploy/receiver.service +36 -0
  5. package/deploy/worker-env-wrapper.cmd +50 -0
  6. package/deploy/worker-env-wrapper.sh +63 -0
  7. package/deploy/worker.service +55 -0
  8. package/package.json +83 -0
  9. package/src/azure-auth.mjs +61 -0
  10. package/src/azure-host.mjs +236 -0
  11. package/src/azure-identity.mjs +63 -0
  12. package/src/azure-prompt.mjs +118 -0
  13. package/src/branch.mjs +80 -0
  14. package/src/budget.mjs +179 -0
  15. package/src/cli.mjs +208 -0
  16. package/src/config.mjs +329 -0
  17. package/src/connection.mjs +40 -0
  18. package/src/cron.mjs +94 -0
  19. package/src/docker-run.mjs +119 -0
  20. package/src/doctor.mjs +1127 -0
  21. package/src/env-allowlist.mjs +198 -0
  22. package/src/env-file.mjs +153 -0
  23. package/src/exit-code.mjs +32 -0
  24. package/src/flow-gate.mjs +82 -0
  25. package/src/forgejo-auth.mjs +77 -0
  26. package/src/forgejo-host.mjs +172 -0
  27. package/src/forgejo-identity.mjs +74 -0
  28. package/src/forgejo-prompt.mjs +123 -0
  29. package/src/forges.mjs +148 -0
  30. package/src/get-token.mjs +226 -0
  31. package/src/git-dirty.mjs +16 -0
  32. package/src/github-app-setup.mjs +517 -0
  33. package/src/github-host.mjs +159 -0
  34. package/src/github-prompt.mjs +286 -0
  35. package/src/gitlab-auth.mjs +72 -0
  36. package/src/gitlab-host.mjs +200 -0
  37. package/src/gitlab-identity.mjs +61 -0
  38. package/src/gitlab-prompt.mjs +123 -0
  39. package/src/identity.mjs +57 -0
  40. package/src/image-preflight.mjs +180 -0
  41. package/src/import-pi.mjs +451 -0
  42. package/src/index.mjs +177 -0
  43. package/src/init.mjs +77 -0
  44. package/src/job-id.mjs +100 -0
  45. package/src/materialize.mjs +138 -0
  46. package/src/outbox.mjs +179 -0
  47. package/src/packages.mjs +188 -0
  48. package/src/pause-windows.mjs +218 -0
  49. package/src/prepare-github.mjs +260 -0
  50. package/src/prepare-local.mjs +76 -0
  51. package/src/prepare.mjs +199 -0
  52. package/src/pricing.mjs +168 -0
  53. package/src/processor.mjs +360 -0
  54. package/src/queue.mjs +152 -0
  55. package/src/run-container.mjs +133 -0
  56. package/src/run-history.mjs +534 -0
  57. package/src/runtime-settings.mjs +188 -0
  58. package/src/sandbox-cli.mjs +156 -0
  59. package/src/sandbox-store.mjs +269 -0
  60. package/src/sandbox.mjs +171 -0
  61. package/src/scheduler-stall-guard.mjs +67 -0
  62. package/src/schedules.mjs +62 -0
  63. package/src/service.mjs +677 -0
  64. package/src/session-key.mjs +108 -0
  65. package/src/session-store.mjs +249 -0
  66. package/src/start.mjs +502 -0
  67. package/src/subscriptions.mjs +208 -0
  68. package/src/triggers.mjs +491 -0
  69. package/src/up.mjs +315 -0
@@ -0,0 +1,360 @@
1
+ import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
2
+ import { configError } from "./config.mjs";
3
+ import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
4
+
5
+ /**
6
+ * The job orchestration. Deliberately a pure-ish function over INJECTED side-effecting deps, so
7
+ * the money-safety ORDER can be tested without GitHub, Docker, or Redis.
8
+ *
9
+ * The order is the contract, and every step before `runContainer` must be free of provider spend:
10
+ *
11
+ * 0. refuse a job image this host does not have -- INT-CONTAINER-RUNTIME-CONTRACT
12
+ * 1. mint a scoped token (GitHub jobs, and local jobs opted in via `github: true`)
13
+ * -- CONST-TOKEN-SCOPED-PER-JOB
14
+ * 2. REFUSE an unprotected default branch (GitHub jobs only -- a local job has no repo)
15
+ * -- REQ-BRANCH-PROTECTION-PRECONDITION
16
+ * 3. resolve the default-branch SHA (fresh API), clone at it, materialise .pi/, write the prompt
17
+ * 4. reserve a budget slot -- CONST-BUDGET-BEFORE-TOKENS
18
+ * 5. ONLY NOW run the container (the only step that spends provider tokens)
19
+ * 6. map the container exit code to retry-vs-success
20
+ *
21
+ * Budget is reserved as late as possible but strictly before the container, so a refusal from an
22
+ * earlier free gate (unprotected repo, clone failure) never consumes a daily slot. The container
23
+ * is the only thing that spends money, so "before tokens" means "before this line".
24
+ *
25
+ * Returns a result object on a non-retryable outcome; THROWS on a retryable (infra) one so BullMQ
26
+ * retries per `attempts`. The caller (the BullMQ processor) turns the thrown/returned distinction
27
+ * into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
28
+ */
29
+ export async function runJob(job, deps) {
30
+ const {
31
+ redis,
32
+ caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
33
+ softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
34
+ tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
35
+ recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
36
+ // (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
37
+ // this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
38
+ // omits it behaves exactly as before -- the container's own failure stays the backstop.
39
+ imagePreflight = async () => ({ ok: true }),
40
+ // (session, { piVersion }) => { promoted, reason, bytes }. Promotes this job's transcript back into
41
+ // the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
42
+ // it behaves exactly as before -- no store, no promotion, no session in the record.
43
+ promoteSession = () => null,
44
+ // (job) => scoped short-lived token. Takes the JOB, not the repo: which forge mints -- and therefore
45
+ // which credential the container gets -- is a property of `job.kind`, and only the wiring knows the
46
+ // map. Called for forge-backed jobs and for local jobs opted in via `github: true`; unflagged local
47
+ // jobs never mint (token stays null).
48
+ mintToken,
49
+ isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
50
+ prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
51
+ // runContainer({ job, token, prepared, name, signal }) => { code, aborted, turns, tokens, session, usage }. It MUST honour
52
+ // `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
53
+ // true at entry (the timeout can fire during a slow prepare). The wiring injects name + signal.
54
+ runContainer,
55
+ cleanup, // (dirs) => void
56
+ comment, // (job, text) => void (issue status; no-op for local jobs)
57
+ log = () => {},
58
+ // The outbox chain collector (INT-OUTBOX-CONTRACT). No-op default so a job whose wiring omits it --
59
+ // or a github job with no /outbox -- chains nothing. It NEVER throws (outbox.mjs), so its counts are
60
+ // additive telemetry that can never flip the parent's completed outcome (CONST-RETRY-INFRA-ONLY).
61
+ collectChain = async () => ({ enqueued: 0, refused: 0 }),
62
+ now = new Date(),
63
+ } = deps;
64
+
65
+ // "Forge-backed" is the negation of local, not an enumeration of forges: a job that is not editing a
66
+ // folder on this host is working against a remote, and every gate below applies for the same reason
67
+ // regardless of WHICH remote. Written this way so a new forge inherits the gates rather than having to
68
+ // be added to them -- the failure mode of an enumeration is a forge that silently skips a money gate.
69
+ const isForgeBacked = job.kind !== "local";
70
+ // A local job opted in via `github: true` (cron trigger opt-in, INT-TRIGGERS-FILE-CONTRACT) mints the
71
+ // same scoped per-job token the github path mints (CONST-TOKEN-SCOPED-PER-JOB). Unflagged local jobs
72
+ // stay tokenless, exactly as before.
73
+ const wantsForgeToken = isForgeBacked || job.github === true;
74
+ let token = null;
75
+ let prepared = null;
76
+ let reserved = false;
77
+
78
+ try {
79
+ // The job image must exist on THIS host before anything else happens. Free, determinate and
80
+ // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
81
+ // image refuses without minting a credential it will not use, cloning a repo it will not read, or
82
+ // burning a cap slot. Jobs run with --pull=never (docker-run.mjs), so an absent image is never
83
+ // fetched -- this refusal IS the whole diagnosis, not a race with a background pull.
84
+ const img = await imagePreflight(job);
85
+ // The image's declared pi version, read on the inspect the preflight already ran. Needed BEFORE the
86
+ // container starts, because a transcript written by a different pi may hold tool-call arguments the
87
+ // current schema no longer accepts -- so the resume has to be refused, not repaired mid-run. Null
88
+ // when the image declares none, which downstream means "never resume": the safe direction.
89
+ const piVersion = img.piVersion ?? null;
90
+ if (img.missing) {
91
+ await comment(job, `Refused: the job image "${img.missing}" is not present on the worker host. Not run.`);
92
+ log("refused_image_missing", { image: img.missing });
93
+ // The image ref is operator-authored config (PI_JOB_IMAGE), never payload, so naming it is PII-safe
94
+ // -- the same class as `repo` above.
95
+ // exitCode/turns/tokens null and budgetReserved false: refused pre-container AND pre-reserve.
96
+ // provider/model ride every terminal result from here down (INT-RUN-HISTORY-FILE-CONTRACT):
97
+ // runJob's `job` IS the effectiveJob (index.mjs), so these are the HOST-effective,
98
+ // overlay-resolved dispatch facts -- never anything a container printed -- and even a
99
+ // pre-container refusal attributes which (provider, model) it was dispatched for. There is
100
+ // deliberately NO `usage` key on the pre-container branches: no run, no ledger, and
101
+ // buildRecord defaults the absent field to null.
102
+ return { outcome: "policy", reason: "job-image-missing", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
103
+ }
104
+ if (img.forgeUnsupported) {
105
+ // The image is present and says it cannot serve this forge -- it ships no CLI for it. Determinate,
106
+ // so a refusal rather than a retry, and pre-spend, because the alternative is a paid container
107
+ // that fails at step 3 on every single delivery with nothing to distinguish it from a bad run.
108
+ //
109
+ // The message names the LIKELY CAUSE rather than the label that detected it: a trigger that
110
+ // forgot `run.image`. The failure is upstream of the thing that noticed it, and an operator
111
+ // reading "the image does not declare azure" has further to walk than one reading "set run.image".
112
+ await comment(
113
+ job,
114
+ `Refused: the job image "${img.forgeUnsupported}" does not support ${img.kind} jobs (it declares: ${img.declared.join(", ")}). Set this trigger's \`run.image\` to an image that does. Not run.`,
115
+ );
116
+ log("refused_image_forge_unsupported", { image: img.forgeUnsupported, kind: img.kind, declared: img.declared });
117
+ return { outcome: "policy", reason: "job-image-forge-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
118
+ }
119
+ if (img.replicaUnsupported) {
120
+ // The image is present and does not declare replica support (REQ-REPLICA-RUNS), so its baked
121
+ // HARD_RULES.md predates the amendment and still hard-codes `pi/issue-<n>` as a SYSTEM rule --
122
+ // which the model treats as authoritative over the user prompt naming `pi/issue-<n>-r2`. Both
123
+ // replicas would push to one branch: not an error, just the push race the feature exists to
124
+ // avoid, with two runs billed and one pull request to show for it.
125
+ //
126
+ // Determinate, so a refusal rather than a retry, and pre-spend, because no version of this gets
127
+ // better by running. Like the forge branch above, the message names the FIX rather than the label
128
+ // that noticed it -- an operator reading "rebuild the image" is already where they need to be.
129
+ await comment(
130
+ job,
131
+ `Refused: the job image "${img.replicaUnsupported}" does not declare replica support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its baked guardrails would name the wrong branch. Rebuild the image from a version that has this feature. Not run.`,
132
+ );
133
+ log("refused_image_replicas_unsupported", { image: img.replicaUnsupported, declared: img.declared });
134
+ return { outcome: "policy", reason: "job-image-replicas-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
135
+ }
136
+ if (img.unavailable) {
137
+ // docker itself did not answer -- transient infra, NOT a determinate refusal. THROWN so BullMQ
138
+ // retries (CONST-RETRY-INFRA-ONLY). `container-never-started` is literally true here, and it reuses
139
+ // the refund path below: a no-op pre-reserve, and still honest if this gate ever moves.
140
+ // provider/model attribute even this pre-container death; no usage -- nothing ran to emit one.
141
+ throw new InfraRetry("docker unavailable, image preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
142
+ }
143
+
144
+ if (wantsForgeToken) {
145
+ token = await mintToken(job);
146
+
147
+ // Defense-in-depth at the DI seam: mintToken is injected, so we cannot assume it routed
148
+ // through get-token's own empty-token guard. An empty credential here would reach
149
+ // env-allowlist's `if (githubToken)` as a falsy value -> GITHUB_TOKEN omitted -> an
150
+ // anonymous paid run. Refuse before reserveBudget so a bad token burns no cap slot.
151
+ if (typeof token !== "string" || token.trim() === "") {
152
+ throw configError("mintToken returned an empty credential");
153
+ }
154
+ }
155
+
156
+ if (isForgeBacked) {
157
+ // REQ-BRANCH-PROTECTION-PRECONDITION. The agent's token can merge, so branch protection is the
158
+ // only technical barrier to a self-merge. Refuse before spending anything. Forge-backed jobs
159
+ // only: a local job has no remote branch to protect.
160
+ if (!(await isDefaultBranchProtected(job, token))) {
161
+ await comment(job, "Refused: the default branch is not protected. See SECURITY.md.");
162
+ log("refused_unprotected", { repo: job.repo });
163
+ // exitCode/turns/tokens null: refused pre-container, so no container exit, turn, or token count exists.
164
+ return { outcome: "policy", reason: "unprotected-branch", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
165
+ }
166
+ }
167
+
168
+ prepared = await prepareWorkspace(job, token, { piVersion }); // resolves SHA, clones, materialises .pi/, writes prompt
169
+
170
+ // A determinate prepare refusal (e.g. sha-gone: the default branch advanced past the resolved
171
+ // tip) is POLICY -- return before reserveBudget so it burns no cap slot and is never retried.
172
+ // Mirrors the branch-protection policy return above. Spread-plus-attribution: the prepare
173
+ // result keeps its own reason and fields, and the host-effective provider/model land beside
174
+ // them exactly as on every other terminal result.
175
+ if (prepared?.outcome === "policy") {
176
+ return { ...prepared, provider: job.provider ?? null, model: job.model ?? null };
177
+ }
178
+
179
+ // Daily TOKEN cap (issue #25): the deliberate check-AFTER control. Token cost is only known
180
+ // post-run, so this cannot check-and-increment before the spend the way the job-count cap does
181
+ // (CONST-BUDGET-BEFORE-TOKENS). It is a read-only GET of prior jobs' recorded spend -- it consumes
182
+ // nothing, so it precedes reserveBudget's INCR and a refusal here burns no job-count slot. It can
183
+ // only stop the NEXT job once the day's accumulated spend has reached the cap; the actual INCRBY
184
+ // happens post-container via recordSpend. Reported before the job-count cap only because both are
185
+ // spend gates; the more-actionable branch-protection precondition is still reported first above --
186
+ // behind only the image check, which outranks it because a missing image blocks EVERY job of EVERY
187
+ // kind on this host, so it is the one the operator must fix first either way.
188
+ const tokenGate = await checkTokenCap(redis, { cap: tokenCap, now });
189
+ if (!tokenGate.allowed) {
190
+ await comment(job, `Over the daily token cap (${tokenGate.spent}/${tokenGate.cap} tokens). Not run.`);
191
+ log("over_token_budget", { spent: tokenGate.spent, cap: tokenGate.cap });
192
+ // budgetReserved false: refused before reserveBudget, so no job-count slot was consumed.
193
+ return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
194
+ }
195
+
196
+ // Budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
197
+ // every active window (day + optional week/month) and the soft-hold band in one atomic pass.
198
+ const budget = await reserveBudget(redis, { caps, softHoldPct, now });
199
+ reserved = true;
200
+ if (!budget.allowed) {
201
+ const w = budget.blockedWindow;
202
+ const win = budget.windows[w];
203
+ if (budget.reason === "soft-hold") {
204
+ await comment(job, `Soft-hold: ${w} spend ${win.reserved}/${win.cap} is inside the ${softHoldPct}% hold band. New starts paused; not run.`);
205
+ log("soft_hold", { window: w, reserved: win.reserved, cap: win.cap, pct: softHoldPct });
206
+ } else {
207
+ await comment(job, `Over the ${w} budget cap (${win.cap}). Not run.`);
208
+ log("over_budget", { window: w, reserved: win.reserved, cap: win.cap });
209
+ }
210
+ // budgetReserved true: the slot is reserved above and kept (a refused reservation still counts). Both
211
+ // over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
212
+ return { outcome: "policy", reason: budget.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
213
+ }
214
+
215
+ const { code, aborted, turns, tokens, session, usage } = await runContainer({ job, token, prepared });
216
+ log("container_exit", { exitCode: code, aborted });
217
+
218
+ // Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
219
+ // so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
220
+ // throws below (an exit-1 container still spent before failing). Record before classifying so all of
221
+ // them are accounted. Only when the cap is enabled (nothing reads the counter otherwise) and the run
222
+ // reported a positive total. NEVER throws: money is already spent, so a Redis blip here must not turn
223
+ // a completed paid job into a failure (mirrors the sink/comment/cleanup fault-isolation posture).
224
+ const tokensSpent = tokens?.total ?? 0;
225
+ if (tokenCap !== null && tokenCap !== undefined && tokensSpent > 0) {
226
+ await recordSpend(redis, tokensSpent, { now }).catch((err) => log("token_spend_error", { reason: err?.message }));
227
+ }
228
+
229
+ // A WORKER-initiated stop (30-min timeout via cancelJob, or graceful-shutdown docker stop) kills
230
+ // the container -> exit 143/137. That is our decision, not an infra fault: it is POLICY and must
231
+ // NOT retry, or a wedged job re-runs into a second PR / double spend. Keyed on the abort FLAG,
232
+ // not the code -- an unbidden 137 (kernel OOM) carries `aborted: false`, falls to the switch, and
233
+ // stays infra-retryable.
234
+ // exitCode/turns/tokens carry the container's own exit, turn count, and usage totals; budgetReserved true post-reserve.
235
+ if (aborted) return { outcome: "policy", reason: "worker-abort", exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
236
+
237
+ switch (code) {
238
+ case EXIT_COMPLETED: {
239
+ // The SOLE chain-collection point. Read the completed parent's /outbox and enqueue children
240
+ // BEFORE the `finally` deletes jobDir -- the await resolves inside this case, so the read
241
+ // finishes before control leaves to cleanup. NOT reached on any other branch (policy, abort,
242
+ // over-budget, infra): an InfraRetry job is retried, so chaining there would double-enqueue.
243
+ // collectChain never throws; chainEnqueued/chainRefused are additive telemetry only.
244
+ const chain = await collectChain({ job, prepared });
245
+ // COMPLETED-ONLY PROMOTION, and the exclusivity is the point rather than an optimisation.
246
+ // A policy or infra exit leaves the canonical transcript byte-identical to what it was
247
+ // before this run, so a retry starts from exactly what the first attempt did -- promote on
248
+ // every exit and "retry" quietly stops meaning re-run and starts meaning continue
249
+ // (CONST-RETRY-INFRA-ONLY). Same completed-only rule INT-OUTBOX-CONTRACT already uses, and
250
+ // it sits beside the chain collection for the same reason: both must happen before the
251
+ // `finally` deletes jobDir. Never throws.
252
+ const promoted = prepared.session ? promoteSession(prepared.session, { piVersion }) : null;
253
+ return {
254
+ outcome: "completed",
255
+ exitCode: code,
256
+ turns,
257
+ tokens,
258
+ // The validated per-model ledger the sink rebuilt off the exit line (parseExitUsage), or
259
+ // null for a fallback-metered or pre-ledger runner. `?? null` keeps the result shape
260
+ // stable under an injected runContainer that predates the field.
261
+ usage: usage ?? null,
262
+ provider: job.provider ?? null,
263
+ model: job.model ?? null,
264
+ session: mergeSession(prepared, session, promoted),
265
+ budgetReserved: true,
266
+ chainEnqueued: chain.enqueued,
267
+ chainRefused: chain.refused,
268
+ };
269
+ }
270
+ case EXIT_POLICY:
271
+ // A policy exit still ran a paid container, so it carries the ledger like the completed
272
+ // branch does -- the spend is real whichever way the runner classified itself.
273
+ return { outcome: "policy", reason: "runner-policy", exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
274
+ case EXIT_INFRA:
275
+ throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
276
+ case 125: // `docker run` itself failed (unusable image reference, bad flag)
277
+ case 126: // the entrypoint exists but is not executable
278
+ case 127: // the entrypoint was not found
279
+ // In all three docker never handed control to the runner, so NOTHING was spent -- which is
280
+ // exactly what `container-never-started` means, and it reuses the refund below rather than
281
+ // keeping a slot the agent never used. These used to fall to `default:`, which kept the slot
282
+ // AND retried, burning a second one. The preflight above converts the KNOWABLE case (an absent
283
+ // image) into a pre-spend policy refusal; a 125 that survives it is a race (the image was
284
+ // removed between the inspect and the run) or a docker-side fault we did not foresee --
285
+ // genuinely infra, and now with a retry that costs nothing.
286
+ throw new InfraRetry(`docker could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
287
+ default:
288
+ throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
289
+ }
290
+ } catch (e) {
291
+ // A spawn fault (docker daemon down / binary missing) reserved a slot but never started a
292
+ // container, so nothing was spent -- give the slot back before the retry. Every other throw
293
+ // here (exit-1 infra, unknown exit) means the container ran and legitimately spent its slot,
294
+ // so `reason` gates the release to the never-started case only. Guarded on `reserved` and run
295
+ // once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
296
+ // budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
297
+ // true for a real container that ran and spent (exit-1 infra / unknown exit).
298
+ if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
299
+ if (reserved && e instanceof InfraRetry && e.reason === "container-never-started") {
300
+ await releaseBudget(redis, { caps, now });
301
+ }
302
+ throw e;
303
+ } finally {
304
+ if (prepared) await cleanup(prepared).catch(() => {});
305
+ }
306
+ }
307
+
308
+ /** Thrown for the retryable (infra) class only. The BullMQ processor lets this propagate to retry. */
309
+ /**
310
+ * The one `session` object the run record carries, from the host's intent and the container's report
311
+ * (INT-RUN-HISTORY-FILE-CONTRACT).
312
+ *
313
+ * Both halves matter and neither is sufficient. The host knows whether a key resolved and which gate
314
+ * refused; only the container knows what pi actually did with the file it was handed. A host that staged
315
+ * a transcript while the runner reports `resumed: false` is a real event -- a corrupt file, a degrade --
316
+ * and with one number alone it is indistinguishable from an ordinary cold start.
317
+ *
318
+ * The runner's verdict WINS on `resumed`, because it is the one that observed the outcome. The host's
319
+ * reason is kept when the runner has none to give (a container that died before its exit line).
320
+ *
321
+ * PII-free by construction: a boolean, a fixed enum, an integer. The key and the branch name are
322
+ * deliberately absent -- this record holds no attacker-chosen string, and a branch name is one.
323
+ */
324
+ function mergeSession(prepared, fromRunner, promoted = null) {
325
+ const host = prepared?.session;
326
+ if (!host && !fromRunner) return null;
327
+ return {
328
+ resumed: fromRunner ? fromRunner.resumed : false,
329
+ // A promotion that was refused is the more useful reason to surface: "locked" or
330
+ // "not-a-regular-file" says why the NEXT run will cold-start, which is the thing an operator
331
+ // chasing "it never resumes" needs. It only ever replaces a reason on the completed path.
332
+ reason: (promoted && !promoted.promoted ? promoted.reason : null) ?? fromRunner?.reason ?? host?.reason ?? null,
333
+ bytes: promoted?.bytes ?? host?.bytes ?? null,
334
+ };
335
+ }
336
+
337
+ export class InfraRetry extends Error {
338
+ constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved } = {}) {
339
+ super(message, cause ? { cause } : undefined);
340
+ this.name = "InfraRetry";
341
+ this.piDispatchRetry = true;
342
+ this.reason = reason ?? message;
343
+ this.exitCode = exitCode ?? null;
344
+ this.turns = turns ?? null;
345
+ this.tokens = tokens ?? null;
346
+ // A deliberate in-passing repair: the EXIT_INFRA throw has passed `session` since the resume
347
+ // feature landed, but this destructure never read it, so every infra-retry record silently
348
+ // recorded session:null and a degrade seen only on a retried attempt left no trace. Latent
349
+ // because buildRecord's `?? null` made the drop indistinguishable from an honest absence.
350
+ this.session = session ?? null;
351
+ // The usage-ledger trio (INT-RUN-HISTORY-FILE-CONTRACT): carried on the throw path so a
352
+ // catch-path record attributes exactly what the return path would have.
353
+ this.usage = usage ?? null;
354
+ this.provider = provider ?? null;
355
+ this.model = model ?? null;
356
+ this.budgetReserved = budgetReserved ?? null;
357
+ }
358
+ }
359
+
360
+ export { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY };
package/src/queue.mjs ADDED
@@ -0,0 +1,152 @@
1
+ import { Queue } from "bullmq";
2
+ import { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId } from "./job-id.mjs";
3
+ import { targetSeparator } from "./forges.mjs";
4
+
5
+ export const QUEUE = "pi-jobs";
6
+ export { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId };
7
+
8
+ export function makeQueue(connection) {
9
+ return new Queue(QUEUE, { connection });
10
+ }
11
+
12
+ /**
13
+ * Enqueue a local-folder job. Returns the jobId. The data shape is what the processor's runJob
14
+ * consumes (kind/folder/flow/task/provider/model/maxTurns).
15
+ *
16
+ * removeOnComplete keeps the dedup window ~= the retention. Unlike webhooks, local jobs are not
17
+ * redelivered, so a modest window is enough.
18
+ */
19
+ export async function enqueueLocalJob(queue, { folder, flow, task, provider, model, maxTurns, image, chainDepth, parentJobId, jobId, now = new Date() }) {
20
+ const minute = now.toISOString().slice(0, 16); // YYYY-MM-DDTHH:MM -- the dedup window
21
+ // A caller-supplied jobId (the outbox collector's retry-idempotent chainedJobId) wins; otherwise the
22
+ // minute-windowed localJobId is the dedup key.
23
+ const id = jobId ?? localJobId({ folder, flow, task, minute });
24
+ // image/chainDepth/parentJobId land on `data` only when present, so a plain non-chained job's data is
25
+ // byte-identical. `image` is the container image this job runs in (INT-TRIGGERS-FILE-CONTRACT); absent
26
+ // resolves the deployment default at job start, never a value frozen here.
27
+ const data = {
28
+ kind: "local",
29
+ folder,
30
+ flow,
31
+ task,
32
+ provider,
33
+ model,
34
+ maxTurns,
35
+ ...(image !== undefined && { image }),
36
+ ...(chainDepth !== undefined && { chainDepth }),
37
+ ...(parentJobId !== undefined && { parentJobId }),
38
+ };
39
+ await queue.add("local", data, {
40
+ jobId: id,
41
+ attempts: 2,
42
+ backoff: { type: "exponential", delay: 60_000 },
43
+ removeOnComplete: { age: 24 * 3600 },
44
+ removeOnFail: { age: 7 * 24 * 3600 },
45
+ });
46
+ return id;
47
+ }
48
+
49
+ // Coalesces rapid re-label spam; the GUID jobId + 31d retention handle exact redelivery, so this
50
+ // window only needs to absorb burst re-labels, not the full redelivery window.
51
+ const SEMANTIC_WINDOW_MS = 10 * 60 * 1000;
52
+
53
+ /**
54
+ * Enqueue a GitHub-triggered job. Returns the jobId. The data shape is what prepare/runJob consumes
55
+ * for the github kind. No `sha` field: the commit is resolved fresh in prepare (C1), so baking a
56
+ * possibly-stale sha here would only race the branch head.
57
+ *
58
+ * `target` is the discriminated subject of the job -- `{ type:"issue"|"pull_request", number, title,
59
+ * body, ... }` -- built by the receiver's filter from the INT-WEBHOOK-PAYLOAD-SUBSET fields. Its `number`
60
+ * keys the semantic dedup window; GitHub issues and PRs share one per-repo number sequence, so the key is
61
+ * collision-free without encoding the type. That is a fact about GitHub, not about forges -- see
62
+ * `enqueueGitLabJob`, where they are separate sequences and the type has to be in the key.
63
+ *
64
+ * Two dedup layers, ADDITIVE and independent:
65
+ * - `jobId` (the delivery GUID) is exact-per-delivery: a redelivered webhook resolves to the same
66
+ * id and BullMQ's `EXISTS jobId` rejects it -- REQ-DEDUP-BY-DELIVERY-GUID.
67
+ * - `deduplication` keys on `repo#number:flow` for SEMANTIC_WINDOW_MS: distinct GUIDs from rapid
68
+ * re-labels or repeated PR pushes coalesce to one active job. It coexists with jobId; it does not
69
+ * replace it.
70
+ */
71
+ export async function enqueueGitHubJob(queue, fields) {
72
+ return await enqueueForgeJob(queue, "github", fields);
73
+ }
74
+
75
+ /**
76
+ * Enqueue a GitLab-triggered job. Structurally the twin of `enqueueGitHubJob` -- and now literally the
77
+ * same body, because everything that differs between them turned out to be two table entries.
78
+ *
79
+ * `projectId` rides the data because every GitLab API path the worker needs takes the numeric project id.
80
+ * A GitLab project path is `group/subgroup/project` with no fixed segment count, so the `owner/name` split
81
+ * the GitHub path uses does not merely fail on one, it SUCCEEDS wrongly: both halves come back non-empty
82
+ * and the project silently becomes its own parent group. Carrying the id sidesteps the grammar entirely;
83
+ * `repo` stays as the human-readable label for logs, run history and pause-window scopes.
84
+ */
85
+ export async function enqueueGitLabJob(queue, fields) {
86
+ return await enqueueForgeJob(queue, "gitlab", fields);
87
+ }
88
+
89
+ /**
90
+ * Enqueue a forge-triggered job of any kind. Returns the jobId.
91
+ *
92
+ * The two named wrappers above are spellings of this. They were separate bodies until a third and fourth
93
+ * forge made that four copies of the retention window, the retry policy, the backoff and BOTH dedup
94
+ * layers -- four places for one of them to be quietly weakened while every test stayed green.
95
+ *
96
+ * The semantic dedup key encodes the TARGET TYPE through `targetSeparator`, which GitHub alone does not
97
+ * need: it numbers issues and pull requests from one per-repo sequence, so `repo#7` names exactly one
98
+ * thing. GitLab numbers them separately, so issue #5 and merge request !5 would collide on `project#5:flow`
99
+ * -- one silently coalescing into the other's 10-minute window and never running. The separator is each
100
+ * forge's own notation, and it lives in the table because it is a fact about the forge.
101
+ *
102
+ * Forge-specific data fields are listed EXPLICITLY rather than collected with a rest spread. A spread
103
+ * would persist whatever a caller happened to pass into durable job data, and this object is copied
104
+ * verbatim into `/job/event.json` -- a place where an unreviewed field has no business.
105
+ *
106
+ * REPLICAS (REQ-REPLICA-RUNS) are the one case where one delivery becomes more than one job, and BOTH dedup
107
+ * layers have to be told, not just the id. The caller loops and passes `replica` 1..N; each pass is an
108
+ * ordinary enqueue with a distinct id and a distinct semantic key. The `replica` suffix on the dedup id is
109
+ * added ONLY when a replica is set, so re-deliveries of each replica still coalesce within the 10-minute
110
+ * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
111
+ * has always been.
112
+ */
113
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, trigger, provider, model, maxTurns, packages, image, resume, replica, replicas }) {
114
+ const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
115
+ // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
116
+ // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
117
+ // only when the filter resolved one, exactly like chainDepth/parentJobId above, so an unflagged trigger's
118
+ // job data is byte-identical. Both sit at JOB level, never inside `trigger` -- that object is descriptive
119
+ // and is copied verbatim into /job/event.json, where an execution knob has no business.
120
+ const data = {
121
+ kind,
122
+ repo,
123
+ ...(projectId !== undefined && { projectId }),
124
+ // Azure's org/project/repository triple, alongside the human-readable `repo` -- the same split gitlab
125
+ // makes with `projectId`, and for the same reason: every Azure API path takes ids and names this label
126
+ // cannot be reassembled into without guessing.
127
+ ...(azure !== undefined && { azure }),
128
+ target,
129
+ flow,
130
+ trigger,
131
+ provider,
132
+ model,
133
+ maxTurns,
134
+ ...(packages !== undefined && { packages }),
135
+ ...(image !== undefined && { image }),
136
+ ...(resume !== undefined && { resume }),
137
+ // Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
138
+ // exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
139
+ // both are integers, so the run record they land in stays PII-free by construction.
140
+ ...(replica !== undefined && { replica }),
141
+ ...(replicas !== undefined && { replicas }),
142
+ };
143
+ await queue.add(kind, data, {
144
+ jobId,
145
+ deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${flow}${replica !== undefined ? `:r${replica}` : ""}`, ttl: SEMANTIC_WINDOW_MS }, // ttl in ms
146
+ attempts: 2,
147
+ backoff: { type: "exponential", delay: 60_000 },
148
+ removeOnComplete: { age: 31 * 24 * 3600 }, // age in seconds -- do not cross units with the ms ttl above
149
+ removeOnFail: { age: 31 * 24 * 3600 },
150
+ });
151
+ return jobId;
152
+ }
@@ -0,0 +1,133 @@
1
+ import { spawn } from "node:child_process";
2
+ import { buildDockerRunArgs, CONTAINER_SESSION_FILE } from "./docker-run.mjs";
3
+ import { buildContainerEnv } from "./env-allowlist.mjs";
4
+ import { resolveJobImage } from "./image-preflight.mjs";
5
+ import { InfraRetry } from "./processor.mjs";
6
+
7
+ /**
8
+ * The real `runContainer` the processor injects. Launches one job container and returns
9
+ * `{ code, aborted, turns, tokens, session, usage }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
10
+ * the 30-min timeout or graceful shutdown), which the processor classifies as POLICY (no retry) per
11
+ * INT-RUNNER-EXIT-CODE-PROTOCOL. The numeric `code` alone cannot say this: a worker SIGKILL and a
12
+ * kernel OOM both surface as 137, so the abort FLAG -- not the code -- is the discriminator.
13
+ *
14
+ * `spawn` (not execFile) because a non-zero exit is NORMAL here: exit 1 (infra) and 2 (policy) are
15
+ * expected outcomes, not errors to reject on. The exit code comes from the `close` event.
16
+ *
17
+ * The container is stopped on abort by the worker wiring (index.mjs onAbort -> docker stop), which
18
+ * causes `docker run` to exit and this promise to resolve. We only handle the entry case here: if
19
+ * the signal is ALREADY aborted (the 30-min timeout fired during a slow prepare), do not start a
20
+ * container at all.
21
+ *
22
+ * Output is streamed to `onOutput` (default: the worker's stdout) so the operator watches the agent
23
+ * work on their own machine -- the natural local UX. When raw capture is enabled
24
+ * (`PI_CAPTURE_JOB_LOGS`), the same output is tee'd to a host-only, gitignored `logs/<jobId>.log`
25
+ * that is never mounted into the container and may contain agent-echoed issue text (PII). The
26
+ * worker's event log and the `.json` status record stay id-only.
27
+ */
28
+ export function makeRunContainer({
29
+ image, // the DEPLOYMENT default (PI_JOB_IMAGE); a trigger's own run.image overrides it per job
30
+ hostEnv = process.env,
31
+ onOutput = (c) => process.stdout.write(c),
32
+ openJobLog = () => ({ write() {}, close: async () => ({ turns: null, tokens: null, session: null, usage: null }) }),
33
+ spawnFn = spawn,
34
+ globalPiDir = null, // REQ-GLOBAL-PI-OVERLAY: operator's global pi overlay dir, mounted :ro; null = off
35
+ allowGlobalExtensions = true, // REQ-GLOBAL-PI-OVERLAY: the staged overlay's extensions load unless PI_GLOBAL_ALLOW_EXTENSIONS=0
36
+ packagePaths = [], // REQ-GLOBAL-PI-OVERLAY: container paths of the operator-staged packages, resolved once at boot
37
+ forwardEnv = [],
38
+ authFromPi = false, // fall back to ~/.pi/agent/auth.json for the provider key when the env has none
39
+ forgeHosts = {}, // per-forge self-hosted instance URLs, so a forge CLI in the container talks to the right one
40
+ }) {
41
+ // async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
42
+ // a rejection, uniformly awaitable by the processor and by tests.
43
+ return async function runContainer({ job, token, prepared, name, signal }) {
44
+ if (signal?.aborted) return { code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null }; // killed before it could start
45
+
46
+ // Closed env allowlist: only the provider key + the declared PI_* vars. Throws (config) if
47
+ // the provider is unconfigured -- the processor turns that into a pre-spend refusal.
48
+ const env = buildContainerEnv({
49
+ provider: job.provider,
50
+ model: job.model,
51
+ maxTurns: job.maxTurns,
52
+ maxTokens: job.maxTokens, // optional per-job token budget (issue #25); undefined => runner meter only
53
+ jobId: name,
54
+ githubToken: token ?? undefined,
55
+ // Which forge minted it, so the token lands in that forge's own variable names and no other.
56
+ forgeKind: job?.kind,
57
+ forgeHosts,
58
+ hostEnv,
59
+ allowGlobalExtensions, // REQ-GLOBAL-PI-OVERLAY: false emits the explicit PI_GLOBAL_ALLOW_EXTENSIONS=0 opt-out
60
+ // REQ-GLOBAL-PI-OVERLAY: the per-job value comes off `job` (like maxTurns), the staged set off
61
+ // the closure (like allowGlobalExtensions) -- so a trigger can withhold what the operator staged.
62
+ // `!== false`, because staged packages LOAD unless a trigger explicitly opts out
63
+ // (INT-TRIGGERS-FILE-CONTRACT). The strictness that used to live in this `=== true` did not
64
+ // disappear, it moved: parseTriggers refuses any non-boolean run.packages fail-loud at load, so a
65
+ // hand-edited string "false" never becomes job data this comparison could misread as an opt-out.
66
+ packagePaths: job.packages === false ? [] : packagePaths,
67
+ forwardEnv, // extra host var names to forward (e.g. a custom provider's key)
68
+ // REQ-RESUMABLE-SESSION: the fixed container path, emitted only when this job HAS a transcript.
69
+ // The constant is imported rather than re-typed so the mount below and this variable name one
70
+ // path -- two literals is how they drift with both suites green.
71
+ sessionFile: prepared.session ? CONTAINER_SESSION_FILE : undefined,
72
+ authFromPi, // source the provider key from pi's auth.json when the env has none
73
+ });
74
+
75
+ const args = buildDockerRunArgs({
76
+ // Same split as packagePaths above: the per-job value off `job`, the deployment value off the closure,
77
+ // so a trigger can name its own toolchain (INT-TRIGGERS-FILE-CONTRACT). Resolved through the SAME
78
+ // function the pre-spend preflight uses (image-preflight.mjs), so the tag that was checked is the tag
79
+ // that runs -- one answer by construction, not two call sites that happen to agree.
80
+ image: resolveJobImage(job, image),
81
+ env,
82
+ jobDir: prepared.jobDir,
83
+ workspace: prepared.workspace,
84
+ outboxDir: prepared.outboxDir, // undefined for github jobs -> docker-run's guard skips the /outbox mount
85
+ // The job's OWN copy, under jobDir -- never the shared store. Undefined when the trigger did not
86
+ // arm run.resume or no key resolved, and docker-run's guard then skips the mount entirely.
87
+ sessionDir: prepared.session?.hostDir,
88
+ globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
89
+ name,
90
+ });
91
+
92
+ // Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
93
+ // sanitizes internally. No container mount, no env var -- the sink lives on this side only.
94
+ const sink = openJobLog(name);
95
+
96
+ return await new Promise((resolve, reject) => {
97
+ const child = spawnFn("docker", args, { stdio: ["ignore", "pipe", "pipe"] });
98
+ // A throwing sink.write is swallowed so a misbehaving sink cannot break the tee or hang the run.
99
+ const tee = (chunk) => {
100
+ onOutput(chunk);
101
+ try {
102
+ sink.write(chunk);
103
+ } catch {}
104
+ };
105
+ child.stdout?.on("data", tee);
106
+ child.stderr?.on("data", tee);
107
+ // docker not found / daemon down -- a transient infra fault, so tag it retryable
108
+ // (CONST-RETRY-INFRA-ONLY). `reason` also cues the processor to release the budget slot,
109
+ // since a container that never started spent nothing.
110
+ child.on("error", (err) => {
111
+ sink.close().catch(() => {}); // best-effort teardown; a rejecting close cannot leak an unhandled rejection
112
+ reject(new InfraRetry("container-never-started", { cause: err, reason: "container-never-started" }));
113
+ });
114
+ child.on("close", async (code) => {
115
+ const aborted = signal?.aborted === true; // capture BEFORE the await
116
+ // A rejecting sink.close is swallowed so a misbehaving sink cannot hang the run; turns/tokens/session/usage fall back to null.
117
+ let turns = null;
118
+ let tokens = null;
119
+ let session = null;
120
+ let usage = null;
121
+ try {
122
+ ({ turns, tokens, session, usage } = await sink.close());
123
+ } catch {
124
+ turns = null;
125
+ tokens = null;
126
+ session = null;
127
+ usage = null;
128
+ }
129
+ resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage } : { code: code ?? 1, aborted: false, turns, tokens, session, usage });
130
+ });
131
+ });
132
+ };
133
+ }