@edgehero/pi-dispatch 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +160 -0
- package/deploy/com.pi-dispatch.worker.plist +66 -0
- package/deploy/nssm-install.cmd +59 -0
- package/deploy/receiver.service +36 -0
- package/deploy/worker-env-wrapper.cmd +50 -0
- package/deploy/worker-env-wrapper.sh +63 -0
- package/deploy/worker.service +55 -0
- package/package.json +83 -0
- package/src/azure-auth.mjs +61 -0
- package/src/azure-host.mjs +236 -0
- package/src/azure-identity.mjs +63 -0
- package/src/azure-prompt.mjs +118 -0
- package/src/branch.mjs +80 -0
- package/src/budget.mjs +179 -0
- package/src/cli.mjs +208 -0
- package/src/config.mjs +329 -0
- package/src/connection.mjs +40 -0
- package/src/cron.mjs +94 -0
- package/src/docker-run.mjs +119 -0
- package/src/doctor.mjs +1127 -0
- package/src/env-allowlist.mjs +198 -0
- package/src/env-file.mjs +153 -0
- package/src/exit-code.mjs +32 -0
- package/src/flow-gate.mjs +82 -0
- package/src/forgejo-auth.mjs +77 -0
- package/src/forgejo-host.mjs +172 -0
- package/src/forgejo-identity.mjs +74 -0
- package/src/forgejo-prompt.mjs +123 -0
- package/src/forges.mjs +148 -0
- package/src/get-token.mjs +226 -0
- package/src/git-dirty.mjs +16 -0
- package/src/github-app-setup.mjs +517 -0
- package/src/github-host.mjs +159 -0
- package/src/github-prompt.mjs +286 -0
- package/src/gitlab-auth.mjs +72 -0
- package/src/gitlab-host.mjs +200 -0
- package/src/gitlab-identity.mjs +61 -0
- package/src/gitlab-prompt.mjs +123 -0
- package/src/identity.mjs +57 -0
- package/src/image-preflight.mjs +180 -0
- package/src/import-pi.mjs +451 -0
- package/src/index.mjs +177 -0
- package/src/init.mjs +77 -0
- package/src/job-id.mjs +100 -0
- package/src/materialize.mjs +138 -0
- package/src/outbox.mjs +179 -0
- package/src/packages.mjs +188 -0
- package/src/pause-windows.mjs +218 -0
- package/src/prepare-github.mjs +260 -0
- package/src/prepare-local.mjs +76 -0
- package/src/prepare.mjs +199 -0
- package/src/pricing.mjs +168 -0
- package/src/processor.mjs +360 -0
- package/src/queue.mjs +152 -0
- package/src/run-container.mjs +133 -0
- package/src/run-history.mjs +534 -0
- package/src/runtime-settings.mjs +188 -0
- package/src/sandbox-cli.mjs +156 -0
- package/src/sandbox-store.mjs +269 -0
- package/src/sandbox.mjs +171 -0
- package/src/scheduler-stall-guard.mjs +67 -0
- package/src/schedules.mjs +62 -0
- package/src/service.mjs +677 -0
- package/src/session-key.mjs +108 -0
- package/src/session-store.mjs +249 -0
- package/src/start.mjs +502 -0
- package/src/subscriptions.mjs +208 -0
- package/src/triggers.mjs +491 -0
- package/src/up.mjs +315 -0
package/src/start.mjs
ADDED
|
@@ -0,0 +1,502 @@
|
|
|
1
|
+
import { execFile } from "node:child_process";
|
|
2
|
+
import { watch } from "node:fs";
|
|
3
|
+
import { dirname, basename } from "node:path";
|
|
4
|
+
import { promisify } from "node:util";
|
|
5
|
+
import { configError, loadConfig } from "./config.mjs";
|
|
6
|
+
import { makeRedisClient, parseConnection } from "./connection.mjs";
|
|
7
|
+
import { reconcile, reloadSchedules } from "./cron.mjs";
|
|
8
|
+
import { makeGitHubAuth } from "./get-token.mjs";
|
|
9
|
+
import { makeGitHubHost } from "./github-host.mjs";
|
|
10
|
+
import { makeGitLabAuth } from "./gitlab-auth.mjs";
|
|
11
|
+
import { makeGitLabHost } from "./gitlab-host.mjs";
|
|
12
|
+
import { makeForgejoAuth } from "./forgejo-auth.mjs";
|
|
13
|
+
import { makeForgejoHost } from "./forgejo-host.mjs";
|
|
14
|
+
import { makeAzureAuth } from "./azure-auth.mjs";
|
|
15
|
+
import { makeAzureHost } from "./azure-host.mjs";
|
|
16
|
+
import { makeImagePreflight } from "./image-preflight.mjs";
|
|
17
|
+
import { createWorker } from "./index.mjs";
|
|
18
|
+
import { makeCollectChain } from "./outbox.mjs";
|
|
19
|
+
import { containerPackagePaths, readStageManifest } from "./packages.mjs";
|
|
20
|
+
import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
|
|
21
|
+
import { listRunningSandboxes } from "./sandbox.mjs";
|
|
22
|
+
import { makeSandboxReaper } from "./sandbox-store.mjs";
|
|
23
|
+
import { makeSessionStore } from "./session-store.mjs";
|
|
24
|
+
import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
|
|
25
|
+
import { makeQueue } from "./queue.mjs";
|
|
26
|
+
import { makeRunContainer } from "./run-container.mjs";
|
|
27
|
+
import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
|
|
28
|
+
import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
|
|
29
|
+
import { loadSchedules } from "./schedules.mjs";
|
|
30
|
+
import { makeStallGuard } from "./scheduler-stall-guard.mjs";
|
|
31
|
+
|
|
32
|
+
const exec = promisify(execFile);
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Boot-time reaper: clear stray `pi-job-*` containers a previous worker crash left behind, before
|
|
36
|
+
* the new worker starts draining. A leaked container keeps spending, so it must go before any new
|
|
37
|
+
* job launches.
|
|
38
|
+
*
|
|
39
|
+
* It runs `docker ps` / `docker rm -f` ONLY. It never inspects a container's exit code, never touches
|
|
40
|
+
* the queue, and never re-enqueues -- queue and retry state belong to Redis, not to docker
|
|
41
|
+
* (INT-RUNNER-EXIT-CODE-PROTOCOL / CONST-RETRY-INFRA-ONLY). It logs container names only (no PII).
|
|
42
|
+
*
|
|
43
|
+
* It assumes ONE worker per docker daemon: a co-located second worker's boot would remove the first's
|
|
44
|
+
* in-flight `pi-job-*` container. That is the accepted v1 shape (DES-CONCURRENCY-3, single worker per
|
|
45
|
+
* host).
|
|
46
|
+
*
|
|
47
|
+
* `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
|
|
48
|
+
* `reaper_skipped`, and boot continues to the worker.
|
|
49
|
+
*/
|
|
50
|
+
/**
|
|
51
|
+
* Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
|
|
52
|
+
* inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
|
|
53
|
+
* `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
|
|
54
|
+
* logs and the worker keeps its boot-time schedulers.
|
|
55
|
+
*/
|
|
56
|
+
function watchTriggersFile(config, queue, log) {
|
|
57
|
+
const path = config.triggersFile;
|
|
58
|
+
const dir = dirname(path) || ".";
|
|
59
|
+
const file = basename(path);
|
|
60
|
+
let timer = null;
|
|
61
|
+
try {
|
|
62
|
+
watch(dir, (_event, changed) => {
|
|
63
|
+
if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
|
|
64
|
+
clearTimeout(timer);
|
|
65
|
+
timer = setTimeout(() => void reloadSchedules(config, queue, { log }), 150);
|
|
66
|
+
}).unref?.();
|
|
67
|
+
log("triggers_watching", { path });
|
|
68
|
+
} catch (err) {
|
|
69
|
+
log("triggers_watch_unavailable", { reason: err?.message });
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
|
|
75
|
+
* and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
|
|
76
|
+
* effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort + unref'd.
|
|
77
|
+
*/
|
|
78
|
+
function watchPauseWindowsFile(config, ref, log) {
|
|
79
|
+
const path = config.pauseWindowsFile;
|
|
80
|
+
const dir = dirname(path) || ".";
|
|
81
|
+
const file = basename(path);
|
|
82
|
+
let timer = null;
|
|
83
|
+
const reload = () => {
|
|
84
|
+
try {
|
|
85
|
+
ref.current = loadPauseWindows(config);
|
|
86
|
+
log("pause_windows_reloaded", { count: ref.current.length });
|
|
87
|
+
} catch (err) {
|
|
88
|
+
log("pause_windows_reload_invalid", { reason: err?.message });
|
|
89
|
+
}
|
|
90
|
+
};
|
|
91
|
+
try {
|
|
92
|
+
watch(dir, (_event, changed) => {
|
|
93
|
+
if (changed && changed !== file) return;
|
|
94
|
+
clearTimeout(timer);
|
|
95
|
+
timer = setTimeout(reload, 150);
|
|
96
|
+
}).unref?.();
|
|
97
|
+
log("pause_windows_watching", { path });
|
|
98
|
+
} catch (err) {
|
|
99
|
+
log("pause_windows_watch_unavailable", { reason: err?.message });
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
export function makeReaper({ log }) {
|
|
104
|
+
return async function reap() {
|
|
105
|
+
try {
|
|
106
|
+
const { stdout } = await exec("docker", ["ps", "--filter", "name=pi-job-", "--format", "{{.Names}}"]);
|
|
107
|
+
const names = stdout
|
|
108
|
+
.split("\n")
|
|
109
|
+
.map((n) => n.trim())
|
|
110
|
+
.filter(Boolean);
|
|
111
|
+
for (const name of names) {
|
|
112
|
+
await exec("docker", ["rm", "-f", name]);
|
|
113
|
+
log("reaped_container", { name });
|
|
114
|
+
}
|
|
115
|
+
} catch (err) {
|
|
116
|
+
log("reaper_skipped", { reason: err?.message });
|
|
117
|
+
}
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* The runnable worker. Reads config, connects to Valkey, wires every REAL dependency the processor
|
|
123
|
+
* needs, and starts draining the queue. `createWorker` already installs the timeout, the
|
|
124
|
+
* abort->docker-stop, and the SIGTERM/SIGINT graceful shutdown.
|
|
125
|
+
*
|
|
126
|
+
* This is where a job's KIND becomes a pair of collaborators. Each forge is one `forges` entry of
|
|
127
|
+
* `{ auth, host }`, and the four deps that used to be bound to one forge -- `mintToken`, `comment`,
|
|
128
|
+
* `isDefaultBranchProtected`, `prepareWorkspace` -- now look their forge up from the job. The processor
|
|
129
|
+
* therefore never branches on which forge a job belongs to; the only place that knows is here.
|
|
130
|
+
*
|
|
131
|
+
* Forge auth is initialised best-effort, per forge: a local-only deployment must still boot when no
|
|
132
|
+
* working GITHUB_AUTH_SOURCE is present, so an auth failure is logged and that forge's deps fail closed
|
|
133
|
+
* per job (mintToken throws configError) rather than blocking startup. Collaborators are injectable
|
|
134
|
+
* (defaulting to the real ones) so the wiring is testable offline with no Redis and no forge.
|
|
135
|
+
*/
|
|
136
|
+
export async function startWorker(
|
|
137
|
+
env = process.env,
|
|
138
|
+
{
|
|
139
|
+
makeAuth = makeGitHubAuth,
|
|
140
|
+
makeHost = makeGitHubHost,
|
|
141
|
+
createWorkerFn = createWorker,
|
|
142
|
+
makeReaper: makeReaperFn = makeReaper,
|
|
143
|
+
makeLogSink: makeLogSinkFn = makeLogSink,
|
|
144
|
+
makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
|
|
145
|
+
makeLogReaper: makeLogReaperFn = makeLogReaper,
|
|
146
|
+
makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
|
|
147
|
+
makeRunContainer: makeRunContainerFn = makeRunContainer,
|
|
148
|
+
makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
|
|
149
|
+
makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
|
|
150
|
+
makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
|
|
151
|
+
makeForgejoAuth: makeForgejoAuthFn = makeForgejoAuth,
|
|
152
|
+
makeForgejoHost: makeForgejoHostFn = makeForgejoHost,
|
|
153
|
+
makeAzureAuth: makeAzureAuthFn = makeAzureAuth,
|
|
154
|
+
makeAzureHost: makeAzureHostFn = makeAzureHost,
|
|
155
|
+
} = {},
|
|
156
|
+
) {
|
|
157
|
+
const config = loadConfig(env);
|
|
158
|
+
const log = (event, fields = {}) => process.stdout.write(`${JSON.stringify({ event, ...fields })}\n`);
|
|
159
|
+
|
|
160
|
+
// DES-CRON-VIA-BULLMQ-SCHEDULER: load and validate the triggers file with the operator present and
|
|
161
|
+
// before any Valkey contact, so a misconfigured schedule refuses startup loudly (configError) rather
|
|
162
|
+
// than upserting a broken scheduler. [] means cron disabled (no PI_TRIGGERS_FILE, or no cron triggers).
|
|
163
|
+
const schedules = loadSchedules(config);
|
|
164
|
+
|
|
165
|
+
// REQ-SCOPED-PAUSE-WINDOWS: load + validate the pause-windows file with the operator present and before any
|
|
166
|
+
// Valkey contact, so a malformed file refuses startup (configError) rather than silently disabling scoped
|
|
167
|
+
// pauses. Held in a mutable ref so the live-reload watcher can hot-swap it. [] means no scoped pauses.
|
|
168
|
+
const pauseWindows = { current: loadPauseWindows(config) };
|
|
169
|
+
|
|
170
|
+
// The forge a job belongs to is resolved PER JOB from `job.kind`, not bound once for the process.
|
|
171
|
+
// Each entry is `{ auth, host }`: `auth` is get-token's `{ mintToken, selfId, source }` (null when that
|
|
172
|
+
// forge is unconfigured or unreachable), `host` is the three methods github-host.mjs returns. The map
|
|
173
|
+
// is the seam -- `forgeFor` below is the only place a kind becomes a pair of collaborators, so the
|
|
174
|
+
// processor never learns which forge it is talking to.
|
|
175
|
+
//
|
|
176
|
+
// Auth stays BEST-EFFORT per forge, exactly as it was: a local-only deployment has no GitHub
|
|
177
|
+
// credentials and must still boot and drain cron jobs. The refusal is deferred to the job that needs
|
|
178
|
+
// the missing credential (the mintToken fallback below), not raised at startup.
|
|
179
|
+
const forges = { github: { auth: null, host: makeHost() } };
|
|
180
|
+
try {
|
|
181
|
+
forges.github.auth = await makeAuth(config.github);
|
|
182
|
+
log("self_identity", { kind: "github", id: forges.github.auth.selfId, source: forges.github.auth.source });
|
|
183
|
+
} catch (err) {
|
|
184
|
+
log("github_auth_unavailable", { kind: "github", reason: err?.message });
|
|
185
|
+
}
|
|
186
|
+
// GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
|
|
187
|
+
// with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
|
|
188
|
+
// answer than an entry that exists and cannot authenticate.
|
|
189
|
+
if (config.gitlab) {
|
|
190
|
+
forges.gitlab = { auth: null, host: makeGitLabHostFn({ apiUrl: config.gitlab.apiUrl }) };
|
|
191
|
+
try {
|
|
192
|
+
forges.gitlab.auth = await makeGitLabAuthFn(config.gitlab);
|
|
193
|
+
log("self_identity", { kind: "gitlab", id: forges.gitlab.auth.selfId, source: forges.gitlab.auth.source });
|
|
194
|
+
} catch (err) {
|
|
195
|
+
log("gitlab_auth_unavailable", { kind: "gitlab", reason: err?.message });
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
// Forgejo joins on the same best-effort terms. Its auth can fail for one reason the others cannot: a
|
|
199
|
+
// repository-scoped token cannot call GET /user, so an operator who scoped their token without setting
|
|
200
|
+
// FORGEJO_BOT_ID lands here. The message names the fix (forgejo-identity.mjs) and the forge stays
|
|
201
|
+
// credential-less, which refuses its jobs at mint time rather than running them unattributed.
|
|
202
|
+
if (config.forgejo) {
|
|
203
|
+
forges.forgejo = { auth: null, host: makeForgejoHostFn({ apiUrl: config.forgejo.apiUrl }) };
|
|
204
|
+
try {
|
|
205
|
+
forges.forgejo.auth = await makeForgejoAuthFn(config.forgejo);
|
|
206
|
+
log("self_identity", { kind: "forgejo", id: forges.forgejo.auth.selfId, source: forges.forgejo.auth.source });
|
|
207
|
+
} catch (err) {
|
|
208
|
+
log("forgejo_auth_unavailable", { kind: "forgejo", reason: err?.message });
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
// Azure joins on the same terms. Its selfId is an OBJECT (`{ id, email }`) rather than a scalar, because
|
|
212
|
+
// a pull-request delivery names an actor by GUID and a work item names them only by address -- the one
|
|
213
|
+
// place a forge's identity does not reduce to a single value.
|
|
214
|
+
if (config.azure) {
|
|
215
|
+
forges.azure = { auth: null, host: makeAzureHostFn({ orgUrl: config.azure.orgUrl }) };
|
|
216
|
+
try {
|
|
217
|
+
forges.azure.auth = await makeAzureAuthFn(config.azure);
|
|
218
|
+
log("self_identity", { kind: "azure", id: forges.azure.auth.selfId?.id ?? null, source: forges.azure.auth.source });
|
|
219
|
+
} catch (err) {
|
|
220
|
+
log("azure_auth_unavailable", { kind: "azure", reason: err?.message });
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** The `{ auth, host }` pair a job's kind names, or `undefined` for a local job (which has no forge). */
|
|
225
|
+
const forgeFor = (job) => forges[job?.kind];
|
|
226
|
+
|
|
227
|
+
// Clear strays left by a previous crash before the worker starts draining. Best-effort: the reaper
|
|
228
|
+
// swallows its own docker errors; this guard keeps any reaper failure from blocking boot.
|
|
229
|
+
try {
|
|
230
|
+
await makeReaperFn({ log })();
|
|
231
|
+
} catch (err) {
|
|
232
|
+
log("reaper_skipped", { reason: err?.message });
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
// REQ-LOCAL-JOB-VISIBILITY: sweep aged `.log`/`.json` history at boot so the logs directory stays
|
|
236
|
+
// bounded across restarts. Best-effort with the same double-wrap posture as the container reaper: the
|
|
237
|
+
// reaper swallows its own fs errors, and this guard keeps any reaper failure from blocking draining.
|
|
238
|
+
try {
|
|
239
|
+
await makeLogReaperFn({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log })();
|
|
240
|
+
} catch (err) {
|
|
241
|
+
log("log_reaper_skipped", { reason: err?.message });
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
|
|
245
|
+
// kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
|
|
246
|
+
// retention policy, a different PII class, and one thing neither sibling needs: it asks docker which
|
|
247
|
+
// sandboxes are live first, because an operator's shell can outlive a worker restart by design and
|
|
248
|
+
// deleting a bind mount underneath it is a confusing failure with a boring cause. Same double-wrap.
|
|
249
|
+
try {
|
|
250
|
+
await makeSandboxReaperFn({
|
|
251
|
+
sandboxDir: config.sandboxDir,
|
|
252
|
+
retentionHours: config.sandboxRetentionHours,
|
|
253
|
+
listRunning: listRunningSandboxes,
|
|
254
|
+
log,
|
|
255
|
+
})();
|
|
256
|
+
} catch (err) {
|
|
257
|
+
log("sandbox_reaper_skipped", { reason: err?.message });
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
// One raw Redis client, shared by the budget (via the worker) and the scheduler stall guard, so it is
|
|
261
|
+
// hoisted out of the createWorkerFn arg object.
|
|
262
|
+
const redis = makeRedisClient(config.valkeyUrl);
|
|
263
|
+
// The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
|
|
264
|
+
// collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
|
|
265
|
+
// Non-failFast: a long-lived handle rides out a Valkey blip. Registered as an extraCloser so shutdown
|
|
266
|
+
// drains it after the worker.
|
|
267
|
+
const runtimeQueue = makeQueue(parseConnection(config.valkeyUrl));
|
|
268
|
+
|
|
269
|
+
// REQ-LOCAL-JOB-VISIBILITY durable run history, all host-side. The raw `.log` sink is gated on
|
|
270
|
+
// captureJobLogs (raw container output is user-authored data, opt-in per no-pii-in-logs); the id-only
|
|
271
|
+
// `.json` record via recordRun is ALWAYS on, so every run leaves a stable, non-PII trace regardless.
|
|
272
|
+
// logsDir wires only into these host-side factories and openJobLog into runContainer -- never into the
|
|
273
|
+
// container env allowlist (no-broad-env-into-container).
|
|
274
|
+
const openJobLog = makeLogSinkFn({ logsDir: config.logsDir, enabled: config.captureJobLogs, log });
|
|
275
|
+
const writeRecord = makeRecordWriterFn({ logsDir: config.logsDir, log });
|
|
276
|
+
// REQ-RESUMABLE-SESSION. Wires into prepareWorkspace and the processor's completed branch only --
|
|
277
|
+
// never into the container env allowlist, exactly as logsDir does not. The one difference from logsDir
|
|
278
|
+
// is that a PER-JOB COPY of one key's transcript IS mounted; the store itself never is.
|
|
279
|
+
const sessionStore = makeSessionStore({
|
|
280
|
+
sessionsDir: config.sessionsDir,
|
|
281
|
+
ttlDays: config.sessionsTtlDays,
|
|
282
|
+
maxBytes: config.sessionMaxBytes,
|
|
283
|
+
log,
|
|
284
|
+
});
|
|
285
|
+
// Boot sweep, beside the log reaper and for the same reason it is beside rather than inside it: these
|
|
286
|
+
// files have a different retention policy and a different PII class. The gate that actually matters is
|
|
287
|
+
// the age check at OPEN -- a worker that never restarts would otherwise resume forever (OQ-007).
|
|
288
|
+
try {
|
|
289
|
+
sessionStore.reapSessions();
|
|
290
|
+
} catch (err) {
|
|
291
|
+
log("session_reaper_skipped", { reason: err?.message });
|
|
292
|
+
}
|
|
293
|
+
const recordRun = ({ job, result, error, startedAt, endedAt }) =>
|
|
294
|
+
writeRecord(buildRecord({ job, result, error, startedAt, endedAt }));
|
|
295
|
+
|
|
296
|
+
// INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
|
|
297
|
+
// closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the eight
|
|
298
|
+
// effective settings from the overlay over env; an invalid overlay returns `{ invalid }` (logged loudly,
|
|
299
|
+
// key-name-only per no-pii-in-logs) so the processor RETURNS a settings-overlay-invalid refusal instead
|
|
300
|
+
// of the run.
|
|
301
|
+
const settingsFile = config.settingsFile;
|
|
302
|
+
const getSettings = () => {
|
|
303
|
+
const res = readOverlay(settingsFile, { log });
|
|
304
|
+
if (res.invalid) {
|
|
305
|
+
log("settings_overlay_invalid", { reason: res.invalid, settingsFile });
|
|
306
|
+
return { invalid: res.invalid };
|
|
307
|
+
}
|
|
308
|
+
return effectiveSettings(config, res.overlay);
|
|
309
|
+
};
|
|
310
|
+
|
|
311
|
+
// Resolve the Worker constructor's slot count once from the overlay: a present overlay may raise or lower
|
|
312
|
+
// boot concurrency. An invalid overlay must NOT dead-end the worker -- fall back to the env/default and let
|
|
313
|
+
// the per-job path enforce the refusal (getSettings already logged the invalid reason).
|
|
314
|
+
const bootSettings = getSettings();
|
|
315
|
+
const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
|
|
316
|
+
|
|
317
|
+
// INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
|
|
318
|
+
// enqueues chained children onto runtimeQueue via enqueueLocalJob -- the same pi-jobs queue the stall
|
|
319
|
+
// guard tears down through. Never throws, so a chain fault cannot flip a completed parent
|
|
320
|
+
// (CONST-RETRY-INFRA-ONLY). The processor calls it as the sole COMPLETED-path chain step.
|
|
321
|
+
const collectChain = makeCollectChain({ queue: runtimeQueue, config, log });
|
|
322
|
+
|
|
323
|
+
// REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest ONCE at boot. The staged set
|
|
324
|
+
// is deploy-time state under the :ro overlay -- identical for every job -- so a per-job re-read would buy
|
|
325
|
+
// nothing and put a filesystem read on the hot path. A missing or unreadable manifest yields [] plus one
|
|
326
|
+
// log line and NEVER a boot failure: a deployment that never opted into packages must not be blocked by
|
|
327
|
+
// it, and `pi-dispatch doctor` is what fails loud on a mismatch between the overlay and the triggers.
|
|
328
|
+
const stagedPackages = config.globalPiDir ? readStageManifest({ globalPiDir: config.globalPiDir }) : null;
|
|
329
|
+
const packagePaths = stagedPackages ? containerPackagePaths(stagedPackages) : [];
|
|
330
|
+
if (config.globalPiDir && !stagedPackages) log("packages_manifest_absent", { overlay: config.globalPiDir });
|
|
331
|
+
|
|
332
|
+
const worker = createWorkerFn({
|
|
333
|
+
connection: parseConnection(config.valkeyUrl),
|
|
334
|
+
concurrency: bootConcurrency,
|
|
335
|
+
getSettings,
|
|
336
|
+
redis,
|
|
337
|
+
recordRun,
|
|
338
|
+
extraClosers: [runtimeQueue],
|
|
339
|
+
// REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
|
|
340
|
+
// Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
|
|
341
|
+
pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
|
|
342
|
+
deps: {
|
|
343
|
+
collectChain,
|
|
344
|
+
// One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
|
|
345
|
+
// image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
|
|
346
|
+
// `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
|
|
347
|
+
// Nothing is memoised: `docker image inspect` costs ~tens of ms against a container run of minutes, and a
|
|
348
|
+
// cache would be wrong in both directions -- an operator who builds the image mid-day would stay refused,
|
|
349
|
+
// one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
|
|
350
|
+
// boot because it is deploy-time state under a :ro mount; the host's image set is not.
|
|
351
|
+
imagePreflight: makeImagePreflightFn({ image: config.jobImage }),
|
|
352
|
+
// Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
|
|
353
|
+
// retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
|
|
354
|
+
promoteSession: sessionStore.promoteSession,
|
|
355
|
+
runContainer: makeRunContainerFn({
|
|
356
|
+
image: config.jobImage,
|
|
357
|
+
hostEnv: env,
|
|
358
|
+
openJobLog,
|
|
359
|
+
globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
|
|
360
|
+
allowGlobalExtensions: config.allowGlobalExtensions,
|
|
361
|
+
packagePaths, // REQ-GLOBAL-PI-OVERLAY: staged package paths; every job receives them unless its trigger set packages:false
|
|
362
|
+
forwardEnv: config.forwardEnv,
|
|
363
|
+
authFromPi: config.authFromPi, // source the provider key from ~/.pi/agent/auth.json when env has none
|
|
364
|
+
// Self-hosted instance URLs, keyed by forge. A MAP rather than one scalar per forge: the table says
|
|
365
|
+
// which variable each lands in, so a forge with no self-hosted concept simply has no entry, and
|
|
366
|
+
// adding one does not widen this signature again.
|
|
367
|
+
forgeHosts: { gitlab: config.gitlab?.apiUrl ?? null, forgejo: config.forgejo?.apiUrl ?? null, azure: config.azure?.orgUrl ?? null },
|
|
368
|
+
}),
|
|
369
|
+
prepareWorkspace: makePrepareWorkspace({
|
|
370
|
+
jobsDir: config.jobsDir,
|
|
371
|
+
forgeFor,
|
|
372
|
+
// REQ-RESURRECTABLE-SANDBOX: the deployment default, resolved per job against run.image so a
|
|
373
|
+
// retained directory records the image that actually ran and a sandbox re-opens that one.
|
|
374
|
+
jobImage: config.jobImage,
|
|
375
|
+
preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
|
|
376
|
+
// The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
|
|
377
|
+
// per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
|
|
378
|
+
findPreviousRun: makeFindPreviousRun({ logsDir: config.logsDir }),
|
|
379
|
+
// Which transcript, if any, a job continues (REQ-RESUMABLE-SESSION). Returns null for every
|
|
380
|
+
// job whose trigger did not arm run.resume, whose key does not resolve, or when
|
|
381
|
+
// PI_SESSIONS_DIR is unset -- and a null means no mount and nothing written.
|
|
382
|
+
resolveSession: sessionStore.resolveSession,
|
|
383
|
+
}),
|
|
384
|
+
// REQ-RESURRECTABLE-SANDBOX. With the window at 0 this IS the old bare `cleanup`, by the same
|
|
385
|
+
// `rm` on the same path -- a deployment that wants no retention keeps today's behaviour exactly.
|
|
386
|
+
cleanup: makeCleanup({ sandboxDir: config.sandboxDir, retentionHours: config.sandboxRetentionHours, log }),
|
|
387
|
+
comment: async (job, text) => {
|
|
388
|
+
// Best-effort: the processor awaits comment() inside its try, so a rejection here would
|
|
389
|
+
// corrupt the job outcome and could drive a wrong retry / second PR (CONST-RETRY-INFRA-ONLY).
|
|
390
|
+
// This adapter NEVER throws.
|
|
391
|
+
const forge = forgeFor(job);
|
|
392
|
+
if (forge?.auth) {
|
|
393
|
+
try {
|
|
394
|
+
const token = await forge.auth.mintToken(job);
|
|
395
|
+
await forge.host.postStatusComment(job, job.target, text, token);
|
|
396
|
+
} catch (err) {
|
|
397
|
+
log("comment_failed", { jobId: job?.id, reason: err?.message });
|
|
398
|
+
}
|
|
399
|
+
return;
|
|
400
|
+
}
|
|
401
|
+
// A local job, or a forge-backed one whose auth never came up. Either way there is nowhere to
|
|
402
|
+
// post, so the line on stdout IS the completion signal (REQ-LOCAL-JOB-VISIBILITY).
|
|
403
|
+
log("comment", { jobId: job?.id, text });
|
|
404
|
+
},
|
|
405
|
+
log,
|
|
406
|
+
// Resolved per job so the credential always comes from the job's OWN forge. A job whose forge has
|
|
407
|
+
// no working auth refuses here, at mint time, rather than running anonymously -- and the refusal
|
|
408
|
+
// names the kind, because with more than one forge configured "auth is broken" is not diagnostic.
|
|
409
|
+
//
|
|
410
|
+
// A LOCAL job reaches this only via the `run.github: true` cron opt-in
|
|
411
|
+
// (INT-TRIGGERS-FILE-CONTRACT), and that flag names github explicitly -- so it mints from the
|
|
412
|
+
// github forge and not from a "default" one. There is deliberately no default: which forge a
|
|
413
|
+
// token comes from must always be something the trigger said.
|
|
414
|
+
mintToken: async (job) => {
|
|
415
|
+
const kind = job?.kind === "local" ? "github" : job?.kind;
|
|
416
|
+
const auth = forges[kind]?.auth;
|
|
417
|
+
if (auth) return await auth.mintToken(job);
|
|
418
|
+
if (kind === "github") {
|
|
419
|
+
throw configError("github jobs and cron triggers with run.github require a working GITHUB_AUTH_SOURCE (gh/pat/app)");
|
|
420
|
+
}
|
|
421
|
+
throw configError(`no forge credentials are configured for job kind ${JSON.stringify(job?.kind)} -- see .env.example`);
|
|
422
|
+
},
|
|
423
|
+
isDefaultBranchProtected: async (job, token) => {
|
|
424
|
+
const host = forgeFor(job)?.host;
|
|
425
|
+
if (!host) throw configError(`no forge host is configured for job kind ${JSON.stringify(job?.kind)}`);
|
|
426
|
+
return await host.isDefaultBranchProtected(job, token);
|
|
427
|
+
},
|
|
428
|
+
},
|
|
429
|
+
});
|
|
430
|
+
|
|
431
|
+
// REQ-LOCAL-JOB-VISIBILITY: exactly one terminal line per job, carrying the job id and outcome,
|
|
432
|
+
// where the operator is already looking. This is the local counterpart of the GitHub issue
|
|
433
|
+
// comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
|
|
434
|
+
// what tells a human a run did nothing. The container's own output already streams via
|
|
435
|
+
// runContainer's onOutput during the run.
|
|
436
|
+
// `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
|
|
437
|
+
// job-image-missing), never
|
|
438
|
+
// user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
|
|
439
|
+
// { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
|
|
440
|
+
worker.on("completed", (job, result) =>
|
|
441
|
+
log("job_completed", { jobId: job?.id, outcome: result?.outcome, ...(result?.reason ? { reason: result.reason } : {}) }),
|
|
442
|
+
);
|
|
443
|
+
worker.on("failed", (job, err) =>
|
|
444
|
+
log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) }),
|
|
445
|
+
);
|
|
446
|
+
|
|
447
|
+
// CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
|
|
448
|
+
// wedged scheduled run is re-paid on every stall. The guard counts stalls per scheduler and tears the
|
|
449
|
+
// scheduler down past the threshold. Keyed on "stalled", not "failed" -- only a stall is the unbounded re-run.
|
|
450
|
+
const guard = makeStallGuard({
|
|
451
|
+
redis,
|
|
452
|
+
threshold: config.schedulerStallMax,
|
|
453
|
+
removeJobScheduler: (id) => runtimeQueue.removeJobScheduler(id),
|
|
454
|
+
log,
|
|
455
|
+
});
|
|
456
|
+
worker.on("stalled", (jobId) => void guard.onStalled(jobId));
|
|
457
|
+
|
|
458
|
+
// DES-CRON-VIA-BULLMQ-SCHEDULER: install the schedule set (and prune orphans) before announcing the
|
|
459
|
+
// worker is up, so schedules_installed always precedes worker_started. An empty set skips the reconcile
|
|
460
|
+
// queue entirely -- no getJobSchedulers Redis hit -- but still logs {0,0} so the operator sees cron is off.
|
|
461
|
+
if (schedules.length > 0) {
|
|
462
|
+
const rq = makeQueue(parseConnection(config.valkeyUrl, { failFast: true }));
|
|
463
|
+
try {
|
|
464
|
+
const r = await reconcile(rq, schedules, { log });
|
|
465
|
+
log("schedules_installed", { installed: r.installed, removed: r.removed });
|
|
466
|
+
} finally {
|
|
467
|
+
await rq.close().catch(() => {});
|
|
468
|
+
}
|
|
469
|
+
} else {
|
|
470
|
+
log("schedules_installed", { installed: 0, removed: 0 });
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
// DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
|
|
474
|
+
// on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
|
|
475
|
+
// Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
|
|
476
|
+
if (config.triggersFile) {
|
|
477
|
+
watchTriggersFile(config, runtimeQueue, log);
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
// REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
|
|
481
|
+
// an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
|
|
482
|
+
if (config.pauseWindowsFile) {
|
|
483
|
+
watchPauseWindowsFile(config, pauseWindows, log);
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
log("worker_started", {
|
|
487
|
+
queue: "pi-jobs",
|
|
488
|
+
concurrency: bootConcurrency, // the slot count the Worker is actually constructed with (overlay may raise/lower it)
|
|
489
|
+
dailyCap: config.dailyCap,
|
|
490
|
+
weeklyCap: config.weeklyCap, // null when the weekly window is disabled
|
|
491
|
+
monthlyCap: config.monthlyCap, // null when the monthly window is disabled
|
|
492
|
+
softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
|
|
493
|
+
image: config.jobImage,
|
|
494
|
+
valkey: config.valkeyUrl,
|
|
495
|
+
logsDir: config.logsDir,
|
|
496
|
+
settingsFile: config.settingsFile,
|
|
497
|
+
captureJobLogs: config.captureJobLogs,
|
|
498
|
+
logRetentionDays: config.logRetentionDays,
|
|
499
|
+
sandboxRetentionHours: config.sandboxRetentionHours, // 0 = retention off; a run's directory is deleted as before
|
|
500
|
+
});
|
|
501
|
+
return worker;
|
|
502
|
+
}
|