@edgehero/pi-dispatch 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.env.example +160 -0
  2. package/deploy/com.pi-dispatch.worker.plist +66 -0
  3. package/deploy/nssm-install.cmd +59 -0
  4. package/deploy/receiver.service +36 -0
  5. package/deploy/worker-env-wrapper.cmd +50 -0
  6. package/deploy/worker-env-wrapper.sh +63 -0
  7. package/deploy/worker.service +55 -0
  8. package/package.json +83 -0
  9. package/src/azure-auth.mjs +61 -0
  10. package/src/azure-host.mjs +236 -0
  11. package/src/azure-identity.mjs +63 -0
  12. package/src/azure-prompt.mjs +118 -0
  13. package/src/branch.mjs +80 -0
  14. package/src/budget.mjs +179 -0
  15. package/src/cli.mjs +208 -0
  16. package/src/config.mjs +329 -0
  17. package/src/connection.mjs +40 -0
  18. package/src/cron.mjs +94 -0
  19. package/src/docker-run.mjs +119 -0
  20. package/src/doctor.mjs +1127 -0
  21. package/src/env-allowlist.mjs +198 -0
  22. package/src/env-file.mjs +153 -0
  23. package/src/exit-code.mjs +32 -0
  24. package/src/flow-gate.mjs +82 -0
  25. package/src/forgejo-auth.mjs +77 -0
  26. package/src/forgejo-host.mjs +172 -0
  27. package/src/forgejo-identity.mjs +74 -0
  28. package/src/forgejo-prompt.mjs +123 -0
  29. package/src/forges.mjs +148 -0
  30. package/src/get-token.mjs +226 -0
  31. package/src/git-dirty.mjs +16 -0
  32. package/src/github-app-setup.mjs +517 -0
  33. package/src/github-host.mjs +159 -0
  34. package/src/github-prompt.mjs +286 -0
  35. package/src/gitlab-auth.mjs +72 -0
  36. package/src/gitlab-host.mjs +200 -0
  37. package/src/gitlab-identity.mjs +61 -0
  38. package/src/gitlab-prompt.mjs +123 -0
  39. package/src/identity.mjs +57 -0
  40. package/src/image-preflight.mjs +180 -0
  41. package/src/import-pi.mjs +451 -0
  42. package/src/index.mjs +177 -0
  43. package/src/init.mjs +77 -0
  44. package/src/job-id.mjs +100 -0
  45. package/src/materialize.mjs +138 -0
  46. package/src/outbox.mjs +179 -0
  47. package/src/packages.mjs +188 -0
  48. package/src/pause-windows.mjs +218 -0
  49. package/src/prepare-github.mjs +260 -0
  50. package/src/prepare-local.mjs +76 -0
  51. package/src/prepare.mjs +199 -0
  52. package/src/pricing.mjs +168 -0
  53. package/src/processor.mjs +360 -0
  54. package/src/queue.mjs +152 -0
  55. package/src/run-container.mjs +133 -0
  56. package/src/run-history.mjs +534 -0
  57. package/src/runtime-settings.mjs +188 -0
  58. package/src/sandbox-cli.mjs +156 -0
  59. package/src/sandbox-store.mjs +269 -0
  60. package/src/sandbox.mjs +171 -0
  61. package/src/scheduler-stall-guard.mjs +67 -0
  62. package/src/schedules.mjs +62 -0
  63. package/src/service.mjs +677 -0
  64. package/src/session-key.mjs +108 -0
  65. package/src/session-store.mjs +249 -0
  66. package/src/start.mjs +502 -0
  67. package/src/subscriptions.mjs +208 -0
  68. package/src/triggers.mjs +491 -0
  69. package/src/up.mjs +315 -0
package/src/start.mjs ADDED
@@ -0,0 +1,502 @@
1
+ import { execFile } from "node:child_process";
2
+ import { watch } from "node:fs";
3
+ import { dirname, basename } from "node:path";
4
+ import { promisify } from "node:util";
5
+ import { configError, loadConfig } from "./config.mjs";
6
+ import { makeRedisClient, parseConnection } from "./connection.mjs";
7
+ import { reconcile, reloadSchedules } from "./cron.mjs";
8
+ import { makeGitHubAuth } from "./get-token.mjs";
9
+ import { makeGitHubHost } from "./github-host.mjs";
10
+ import { makeGitLabAuth } from "./gitlab-auth.mjs";
11
+ import { makeGitLabHost } from "./gitlab-host.mjs";
12
+ import { makeForgejoAuth } from "./forgejo-auth.mjs";
13
+ import { makeForgejoHost } from "./forgejo-host.mjs";
14
+ import { makeAzureAuth } from "./azure-auth.mjs";
15
+ import { makeAzureHost } from "./azure-host.mjs";
16
+ import { makeImagePreflight } from "./image-preflight.mjs";
17
+ import { createWorker } from "./index.mjs";
18
+ import { makeCollectChain } from "./outbox.mjs";
19
+ import { containerPackagePaths, readStageManifest } from "./packages.mjs";
20
+ import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
21
+ import { listRunningSandboxes } from "./sandbox.mjs";
22
+ import { makeSandboxReaper } from "./sandbox-store.mjs";
23
+ import { makeSessionStore } from "./session-store.mjs";
24
+ import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
25
+ import { makeQueue } from "./queue.mjs";
26
+ import { makeRunContainer } from "./run-container.mjs";
27
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
28
+ import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
29
+ import { loadSchedules } from "./schedules.mjs";
30
+ import { makeStallGuard } from "./scheduler-stall-guard.mjs";
31
+
32
+ const exec = promisify(execFile);
33
+
34
+ /**
35
+ * Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
36
+ * the new worker starts draining. A leaked container keeps spending, so it must go before any new
37
+ * job launches.
38
+ *
39
+ * It runs `docker ps` / `docker rm -f` ONLY. It never inspects a container's exit code, never touches
40
+ * the queue, and never re-enqueues -- queue and retry state belong to Redis, not to docker
41
+ * (INT-RUNNER-EXIT-CODE-PROTOCOL / CONST-RETRY-INFRA-ONLY). It logs container names only (no PII).
42
+ *
43
+ * It assumes ONE worker per docker daemon: a co-located second worker's boot would remove the first's
44
+ * in-flight `pi-job-*` container. That is the accepted v1 shape (DES-CONCURRENCY-3, single worker per
45
+ * host).
46
+ *
47
+ * `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
48
+ * `reaper_skipped`, and boot continues to the worker.
49
+ */
50
+ /**
51
+ * Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
52
+ * inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
53
+ * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
54
+ * logs and the worker keeps its boot-time schedulers.
55
+ */
56
+ function watchTriggersFile(config, queue, log) {
57
+ const path = config.triggersFile;
58
+ const dir = dirname(path) || ".";
59
+ const file = basename(path);
60
+ let timer = null;
61
+ try {
62
+ watch(dir, (_event, changed) => {
63
+ if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
64
+ clearTimeout(timer);
65
+ timer = setTimeout(() => void reloadSchedules(config, queue, { log }), 150);
66
+ }).unref?.();
67
+ log("triggers_watching", { path });
68
+ } catch (err) {
69
+ log("triggers_watch_unavailable", { reason: err?.message });
70
+ }
71
+ }
72
+
73
+ /**
74
+ * Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
75
+ * and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
76
+ * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort + unref'd.
77
+ */
78
+ function watchPauseWindowsFile(config, ref, log) {
79
+ const path = config.pauseWindowsFile;
80
+ const dir = dirname(path) || ".";
81
+ const file = basename(path);
82
+ let timer = null;
83
+ const reload = () => {
84
+ try {
85
+ ref.current = loadPauseWindows(config);
86
+ log("pause_windows_reloaded", { count: ref.current.length });
87
+ } catch (err) {
88
+ log("pause_windows_reload_invalid", { reason: err?.message });
89
+ }
90
+ };
91
+ try {
92
+ watch(dir, (_event, changed) => {
93
+ if (changed && changed !== file) return;
94
+ clearTimeout(timer);
95
+ timer = setTimeout(reload, 150);
96
+ }).unref?.();
97
+ log("pause_windows_watching", { path });
98
+ } catch (err) {
99
+ log("pause_windows_watch_unavailable", { reason: err?.message });
100
+ }
101
+ }
102
+
103
+ export function makeReaper({ log }) {
104
+ return async function reap() {
105
+ try {
106
+ const { stdout } = await exec("docker", ["ps", "--filter", "name=pi-job-", "--format", "{{.Names}}"]);
107
+ const names = stdout
108
+ .split("\n")
109
+ .map((n) => n.trim())
110
+ .filter(Boolean);
111
+ for (const name of names) {
112
+ await exec("docker", ["rm", "-f", name]);
113
+ log("reaped_container", { name });
114
+ }
115
+ } catch (err) {
116
+ log("reaper_skipped", { reason: err?.message });
117
+ }
118
+ };
119
+ }
120
+
121
+ /**
122
+ * The runnable worker. Reads config, connects to Valkey, wires every REAL dependency the processor
123
+ * needs, and starts draining the queue. `createWorker` already installs the timeout, the
124
+ * abort->docker-stop, and the SIGTERM/SIGINT graceful shutdown.
125
+ *
126
+ * This is where a job's KIND becomes a pair of collaborators. Each forge is one `forges` entry of
127
+ * `{ auth, host }`, and the four deps that used to be bound to one forge -- `mintToken`, `comment`,
128
+ * `isDefaultBranchProtected`, `prepareWorkspace` -- now look their forge up from the job. The processor
129
+ * therefore never branches on which forge a job belongs to; the only place that knows is here.
130
+ *
131
+ * Forge auth is initialised best-effort, per forge: a local-only deployment must still boot when no
132
+ * working GITHUB_AUTH_SOURCE is present, so an auth failure is logged and that forge's deps fail closed
133
+ * per job (mintToken throws configError) rather than blocking startup. Collaborators are injectable
134
+ * (defaulting to the real ones) so the wiring is testable offline with no Redis and no forge.
135
+ */
136
+ export async function startWorker(
137
+ env = process.env,
138
+ {
139
+ makeAuth = makeGitHubAuth,
140
+ makeHost = makeGitHubHost,
141
+ createWorkerFn = createWorker,
142
+ makeReaper: makeReaperFn = makeReaper,
143
+ makeLogSink: makeLogSinkFn = makeLogSink,
144
+ makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
145
+ makeLogReaper: makeLogReaperFn = makeLogReaper,
146
+ makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
147
+ makeRunContainer: makeRunContainerFn = makeRunContainer,
148
+ makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
149
+ makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
150
+ makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
151
+ makeForgejoAuth: makeForgejoAuthFn = makeForgejoAuth,
152
+ makeForgejoHost: makeForgejoHostFn = makeForgejoHost,
153
+ makeAzureAuth: makeAzureAuthFn = makeAzureAuth,
154
+ makeAzureHost: makeAzureHostFn = makeAzureHost,
155
+ } = {},
156
+ ) {
157
+ const config = loadConfig(env);
158
+ const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields })}\n`);
159
+
160
+ // DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
161
+ // before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
162
+ // than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
163
+ const schedules = loadSchedules(config);
164
+
165
+ // REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
166
+ // Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
167
+ // pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
168
+ const pauseWindows = { current: loadPauseWindows(config) };
169
+
170
+ // The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
171
+ // Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
172
+ // forge is unconfigured or unreachable), `host` is the three methods github-host.mjs returns. The map
173
+ // is the seam -- `forgeFor` below is the only place a kind becomes a pair of collaborators, so the
174
+ // processor never learns which forge it is talking to.
175
+ //
176
+ // Auth stays BEST-EFFORT per forge, exactly as it was: a local-only deployment has no GitHub
177
+ // credentials and must still boot and drain cron jobs. The refusal is deferred to the job that needs
178
+ // the missing credential (the mintToken fallback below), not raised at startup.
179
+ const forges = { github: { auth: null, host: makeHost() } };
180
+ try {
181
+ forges.github.auth = await makeAuth(config.github);
182
+ log("self_identity", { kind: "github", id: forges.github.auth.selfId, source: forges.github.auth.source });
183
+ } catch (err) {
184
+ log("github_auth_unavailable", { kind: "github", reason: err?.message });
185
+ }
186
+ // GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
187
+ // with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
188
+ // answer than an entry that exists and cannot authenticate.
189
+ if (config.gitlab) {
190
+ forges.gitlab = { auth: null, host: makeGitLabHostFn({ apiUrl: config.gitlab.apiUrl }) };
191
+ try {
192
+ forges.gitlab.auth = await makeGitLabAuthFn(config.gitlab);
193
+ log("self_identity", { kind: "gitlab", id: forges.gitlab.auth.selfId, source: forges.gitlab.auth.source });
194
+ } catch (err) {
195
+ log("gitlab_auth_unavailable", { kind: "gitlab", reason: err?.message });
196
+ }
197
+ }
198
+ // Forgejo joins on the same best-effort terms. Its auth can fail for one reason the others cannot: a
199
+ // repository-scoped token cannot call GET /user, so an operator who scoped their token without setting
200
+ // FORGEJO_BOT_ID lands here. The message names the fix (forgejo-identity.mjs) and the forge stays
201
+ // credential-less, which refuses its jobs at mint time rather than running them unattributed.
202
+ if (config.forgejo) {
203
+ forges.forgejo = { auth: null, host: makeForgejoHostFn({ apiUrl: config.forgejo.apiUrl }) };
204
+ try {
205
+ forges.forgejo.auth = await makeForgejoAuthFn(config.forgejo);
206
+ log("self_identity", { kind: "forgejo", id: forges.forgejo.auth.selfId, source: forges.forgejo.auth.source });
207
+ } catch (err) {
208
+ log("forgejo_auth_unavailable", { kind: "forgejo", reason: err?.message });
209
+ }
210
+ }
211
+ // Azure joins on the same terms. Its selfId is an OBJECT (`{ id, email }`) rather than a scalar, because
212
+ // a pull-request delivery names an actor by GUID and a work item names them only by address -- the one
213
+ // place a forge's identity does not reduce to a single value.
214
+ if (config.azure) {
215
+ forges.azure = { auth: null, host: makeAzureHostFn({ orgUrl: config.azure.orgUrl }) };
216
+ try {
217
+ forges.azure.auth = await makeAzureAuthFn(config.azure);
218
+ log("self_identity", { kind: "azure", id: forges.azure.auth.selfId?.id ?? null, source: forges.azure.auth.source });
219
+ } catch (err) {
220
+ log("azure_auth_unavailable", { kind: "azure", reason: err?.message });
221
+ }
222
+ }
223
+
224
+ /** The `{ auth, host }` pair a job's kind names, or `undefined` for a local job (which has no forge). */
225
+ const forgeFor = (job) => forges[job?.kind];
226
+
227
+ // Clear strays left by a previous crash before the worker starts draining. Best-effort: the reaper
228
+ // swallows its own docker errors; this guard keeps any reaper failure from blocking boot.
229
+ try {
230
+ await makeReaperFn({ log })();
231
+ } catch (err) {
232
+ log("reaper_skipped", { reason: err?.message });
233
+ }
234
+
235
+ // REQ-LOCAL-JOB-VISIBILITY: sweep aged `.log`/`.json` history at boot so the logs directory stays
236
+ // bounded across restarts. Best-effort with the same double-wrap posture as the container reaper: the
237
+ // reaper swallows its own fs errors, and this guard keeps any reaper failure from blocking draining.
238
+ try {
239
+ await makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log })();
240
+ } catch (err) {
241
+ log("log_reaper_skipped", { reason: err?.message });
242
+ }
243
+
244
+ // REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
245
+ // kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
246
+ // retention policy, a different PII class, and one thing neither sibling needs: it asks docker which
247
+ // sandboxes are live first, because an operator's shell can outlive a worker restart by design and
248
+ // deleting a bind mount underneath it is a confusing failure with a boring cause. Same double-wrap.
249
+ try {
250
+ await makeSandboxReaperFn({
251
+ sandboxDir: config.sandboxDir,
252
+ retentionHours: config.sandboxRetentionHours,
253
+ listRunning: listRunningSandboxes,
254
+ log,
255
+ })();
256
+ } catch (err) {
257
+ log("sandbox_reaper_skipped", { reason: err?.message });
258
+ }
259
+
260
+ // One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
261
+ // hoisted out of the createWorkerFn arg object.
262
+ const redis = makeRedisClient(config.valkeyUrl);
263
+ // The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
264
+ // collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
265
+ // Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
266
+ // drains it after the worker.
267
+ const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
268
+
269
+ // REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
270
+ // captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
271
+ // `.json` record via recordRun is ALWAYS on, so every run leaves a stable, non-PII trace regardless.
272
+ // logsDir wires only into these host-side factories and openJobLog into runContainer -- never into the
273
+ // container env allowlist (no-broad-env-into-container).
274
+ const openJobLog = makeLogSinkFn({ logsDir: config.logsDir, enabled: config.captureJobLogs, log });
275
+ const writeRecord = makeRecordWriterFn({ logsDir: config.logsDir, log });
276
+ // REQ-RESUMABLE-SESSION. Wires into prepareWorkspace and the processor's completed branch only --
277
+ // never into the container env allowlist, exactly as logsDir does not. The one difference from logsDir
278
+ // is that a PER-JOB COPY of one key's transcript IS mounted; the store itself never is.
279
+ const sessionStore = makeSessionStore({
280
+ sessionsDir: config.sessionsDir,
281
+ ttlDays: config.sessionsTtlDays,
282
+ maxBytes: config.sessionMaxBytes,
283
+ log,
284
+ });
285
+ // Boot sweep, beside the log reaper and for the same reason it is beside rather than inside it: these
286
+ // files have a different retention policy and a different PII class. The gate that actually matters is
287
+ // the age check at OPEN -- a worker that never restarts would otherwise resume forever (OQ-007).
288
+ try {
289
+ sessionStore.reapSessions();
290
+ } catch (err) {
291
+ log("session_reaper_skipped", { reason: err?.message });
292
+ }
293
+ const recordRun = ({ job, result, error, startedAt, endedAt }) =>
294
+ writeRecord(buildRecord({ job, result, error, startedAt, endedAt }));
295
+
296
+ // INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
297
+ // closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the eight
298
+ // effective settings from the overlay over env; an invalid overlay returns `{ invalid }` (logged loudly,
299
+ // key-name-only per no-pii-in-logs) so the processor RETURNS a settings-overlay-invalid refusal instead
300
+ // of the run.
301
+ const settingsFile = config.settingsFile;
302
+ const getSettings = () => {
303
+ const res = readOverlay(settingsFile, { log });
304
+ if (res.invalid) {
305
+ log("settings_overlay_invalid", { reason: res.invalid, settingsFile });
306
+ return { invalid: res.invalid };
307
+ }
308
+ return effectiveSettings(config, res.overlay);
309
+ };
310
+
311
+ // Resolve the Worker constructor's slot count once from the overlay: a present overlay may raise or lower
312
+ // boot concurrency. An invalid overlay must NOT dead-end the worker -- fall back to the env/default and let
313
+ // the per-job path enforce the refusal (getSettings already logged the invalid reason).
314
+ const bootSettings = getSettings();
315
+ const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
316
+
317
+ // INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
318
+ // enqueues chained children onto runtimeQueue via enqueueLocalJob -- the same pi-jobs queue the stall
319
+ // guard tears down through. Never throws, so a chain fault cannot flip a completed parent
320
+ // (CONST-RETRY-INFRA-ONLY). The processor calls it as the sole COMPLETED-path chain step.
321
+ const collectChain = makeCollectChain({ queue: runtimeQueue, config, log });
322
+
323
+ // REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest ONCE at boot. The staged set
324
+ // is deploy-time state under the :ro overlay -- identical for every job -- so a per-job re-read would buy
325
+ // nothing and put a filesystem read on the hot path. A missing or unreadable manifest yields [] plus one
326
+ // log line and NEVER a boot failure: a deployment that never opted into packages must not be blocked by
327
+ // it, and `pi-dispatch doctor` is what fails loud on a mismatch between the overlay and the triggers.
328
+ const stagedPackages = config.globalPiDir ? readStageManifest({ globalPiDir: config.globalPiDir }) : null;
329
+ const packagePaths = stagedPackages ? containerPackagePaths(stagedPackages) : [];
330
+ if (config.globalPiDir && !stagedPackages) log("packages_manifest_absent", { overlay: config.globalPiDir });
331
+
332
+ const worker = createWorkerFn({
333
+ connection: parseConnection(config.valkeyUrl),
334
+ concurrency: bootConcurrency,
335
+ getSettings,
336
+ redis,
337
+ recordRun,
338
+ extraClosers: [runtimeQueue],
339
+ // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
340
+ // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
341
+ pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
342
+ deps: {
343
+ collectChain,
344
+ // One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
345
+ // image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
346
+ // `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
347
+ // Nothing is memoised: `docker image inspect` costs ~tens of ms against a container run of minutes, and a
348
+ // cache would be wrong in both directions -- an operator who builds the image mid-day would stay refused,
349
+ // one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
350
+ // boot because it is deploy-time state under a :ro mount; the host's image set is not.
351
+ imagePreflight: makeImagePreflightFn({ image: config.jobImage }),
352
+ // Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
353
+ // retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
354
+ promoteSession: sessionStore.promoteSession,
355
+ runContainer: makeRunContainerFn({
356
+ image: config.jobImage,
357
+ hostEnv: env,
358
+ openJobLog,
359
+ globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
360
+ allowGlobalExtensions: config.allowGlobalExtensions,
361
+ packagePaths, // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set packages:false
362
+ forwardEnv: config.forwardEnv,
363
+ authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
364
+ // Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
365
+ // which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
366
+ // adding one does not widen this signature again.
367
+ forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
368
+ }),
369
+ prepareWorkspace: makePrepareWorkspace({
370
+ jobsDir: config.jobsDir,
371
+ forgeFor,
372
+ // REQ-RESURRECTABLE-SANDBOX: the deployment default, resolved per job against run.image so a
373
+ // retained directory records the image that actually ran and a sandbox re-opens that one.
374
+ jobImage: config.jobImage,
375
+ preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
376
+ // The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
377
+ // per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
378
+ findPreviousRun: makeFindPreviousRun({ logsDir: config.logsDir }),
379
+ // Which transcript, if any, a job continues (REQ-RESUMABLE-SESSION). Returns null for every
380
+ // job whose trigger did not arm run.resume, whose key does not resolve, or when
381
+ // PI_SESSIONS_DIR is unset -- and a null means no mount and nothing written.
382
+ resolveSession: sessionStore.resolveSession,
383
+ }),
384
+ // REQ-RESURRECTABLE-SANDBOX. With the window at 0 this IS the old bare `cleanup`, by the same
385
+ // `rm` on the same path -- a deployment that wants no retention keeps today's behaviour exactly.
386
+ cleanup: makeCleanup({ sandboxDir: config.sandboxDir, retentionHours: config.sandboxRetentionHours, log }),
387
+ comment: async (job, text) => {
388
+ // Best-effort: the processor awaits comment() inside its try, so a rejection here would
389
+ // corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
390
+ // This adapter NEVER throws.
391
+ const forge = forgeFor(job);
392
+ if (forge?.auth) {
393
+ try {
394
+ const token = await forge.auth.mintToken(job);
395
+ await forge.host.postStatusComment(job, job.target, text, token);
396
+ } catch (err) {
397
+ log("comment_failed", { jobId: job?.id, reason: err?.message });
398
+ }
399
+ return;
400
+ }
401
+ // A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
402
+ // post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
403
+ log("comment", { jobId: job?.id, text });
404
+ },
405
+ log,
406
+ // Resolved per job so the credential always comes from the job's OWN forge. A job whose forge has
407
+ // no working auth refuses here, at mint time, rather than running anonymously -- and the refusal
408
+ // names the kind, because with more than one forge configured "auth is broken" is not diagnostic.
409
+ //
410
+ // A LOCAL job reaches this only via the `run.github: true` cron opt-in
411
+ // (INT-TRIGGERS-FILE-CONTRACT), and that flag names github explicitly -- so it mints from the
412
+ // github forge and not from a "default" one. There is deliberately no default: which forge a
413
+ // token comes from must always be something the trigger said.
414
+ mintToken: async (job) => {
415
+ const kind = job?.kind === "local" ? "github" : job?.kind;
416
+ const auth = forges[kind]?.auth;
417
+ if (auth) return await auth.mintToken(job);
418
+ if (kind === "github") {
419
+ throw configError("github jobs and cron triggers with run.github require a working GITHUB_AUTH_SOURCE (gh/pat/app)");
420
+ }
421
+ throw configError(`no forge credentials are configured for job kind ${JSON.stringify(job?.kind)} -- see .env.example`);
422
+ },
423
+ isDefaultBranchProtected: async (job, token) => {
424
+ const host = forgeFor(job)?.host;
425
+ if (!host) throw configError(`no forge host is configured for job kind ${JSON.stringify(job?.kind)}`);
426
+ return await host.isDefaultBranchProtected(job, token);
427
+ },
428
+ },
429
+ });
430
+
431
+ // REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
432
+ // where the operator is already looking. This is the local counterpart of the GitHub issue
433
+ // comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
434
+ // what tells a human a run did nothing. The container's own output already streams via
435
+ // runContainer's onOutput during the run.
436
+ // `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
437
+ // job-image-missing), never
438
+ // user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
439
+ // { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
440
+ worker.on("completed", (job, result) =>
441
+ log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
442
+ );
443
+ worker.on("failed", (job, err) =>
444
+ log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
445
+ );
446
+
447
+ // CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
448
+ // wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
449
+ // scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
450
+ const guard = makeStallGuard({
451
+ redis,
452
+ threshold: config.schedulerStallMax,
453
+ removeJobScheduler: (id) => runtimeQueue.removeJobScheduler(id),
454
+ log,
455
+ });
456
+ worker.on("stalled", (jobId) => void guard.onStalled(jobId));
457
+
458
+ // DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
459
+ // worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
460
+ // queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
461
+ if (schedules.length > 0) {
462
+ const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }));
463
+ try {
464
+ const r = await reconcile(rq, schedules, { log });
465
+ log("schedules_installed", { installed: r.installed, removed: r.removed });
466
+ } finally {
467
+ await rq.close().catch(() => {});
468
+ }
469
+ } else {
470
+ log("schedules_installed", { installed: 0, removed: 0 });
471
+ }
472
+
473
+ // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
474
+ // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
475
+ // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
476
+ if (config.triggersFile) {
477
+ watchTriggersFile(config, runtimeQueue, log);
478
+ }
479
+
480
+ // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
481
+ // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
482
+ if (config.pauseWindowsFile) {
483
+ watchPauseWindowsFile(config, pauseWindows, log);
484
+ }
485
+
486
+ log("worker_started", {
487
+ queue: "pi-jobs",
488
+ concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
489
+ dailyCap: config.dailyCap,
490
+ weeklyCap: config.weeklyCap, // null when the weekly window is disabled
491
+ monthlyCap: config.monthlyCap, // null when the monthly window is disabled
492
+ softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
493
+ image: config.jobImage,
494
+ valkey: config.valkeyUrl,
495
+ logsDir: config.logsDir,
496
+ settingsFile: config.settingsFile,
497
+ captureJobLogs: config.captureJobLogs,
498
+ logRetentionDays: config.logRetentionDays,
499
+ sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
500
+ });
501
+ return worker;
502
+ }