@edgehero/pi-dispatch 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +160 -0
- package/deploy/com.pi-dispatch.worker.plist +66 -0
- package/deploy/nssm-install.cmd +59 -0
- package/deploy/receiver.service +36 -0
- package/deploy/worker-env-wrapper.cmd +50 -0
- package/deploy/worker-env-wrapper.sh +63 -0
- package/deploy/worker.service +55 -0
- package/package.json +83 -0
- package/src/azure-auth.mjs +61 -0
- package/src/azure-host.mjs +236 -0
- package/src/azure-identity.mjs +63 -0
- package/src/azure-prompt.mjs +118 -0
- package/src/branch.mjs +80 -0
- package/src/budget.mjs +179 -0
- package/src/cli.mjs +208 -0
- package/src/config.mjs +329 -0
- package/src/connection.mjs +40 -0
- package/src/cron.mjs +94 -0
- package/src/docker-run.mjs +119 -0
- package/src/doctor.mjs +1127 -0
- package/src/env-allowlist.mjs +198 -0
- package/src/env-file.mjs +153 -0
- package/src/exit-code.mjs +32 -0
- package/src/flow-gate.mjs +82 -0
- package/src/forgejo-auth.mjs +77 -0
- package/src/forgejo-host.mjs +172 -0
- package/src/forgejo-identity.mjs +74 -0
- package/src/forgejo-prompt.mjs +123 -0
- package/src/forges.mjs +148 -0
- package/src/get-token.mjs +226 -0
- package/src/git-dirty.mjs +16 -0
- package/src/github-app-setup.mjs +517 -0
- package/src/github-host.mjs +159 -0
- package/src/github-prompt.mjs +286 -0
- package/src/gitlab-auth.mjs +72 -0
- package/src/gitlab-host.mjs +200 -0
- package/src/gitlab-identity.mjs +61 -0
- package/src/gitlab-prompt.mjs +123 -0
- package/src/identity.mjs +57 -0
- package/src/image-preflight.mjs +180 -0
- package/src/import-pi.mjs +451 -0
- package/src/index.mjs +177 -0
- package/src/init.mjs +77 -0
- package/src/job-id.mjs +100 -0
- package/src/materialize.mjs +138 -0
- package/src/outbox.mjs +179 -0
- package/src/packages.mjs +188 -0
- package/src/pause-windows.mjs +218 -0
- package/src/prepare-github.mjs +260 -0
- package/src/prepare-local.mjs +76 -0
- package/src/prepare.mjs +199 -0
- package/src/pricing.mjs +168 -0
- package/src/processor.mjs +360 -0
- package/src/queue.mjs +152 -0
- package/src/run-container.mjs +133 -0
- package/src/run-history.mjs +534 -0
- package/src/runtime-settings.mjs +188 -0
- package/src/sandbox-cli.mjs +156 -0
- package/src/sandbox-store.mjs +269 -0
- package/src/sandbox.mjs +171 -0
- package/src/scheduler-stall-guard.mjs +67 -0
- package/src/schedules.mjs +62 -0
- package/src/service.mjs +677 -0
- package/src/session-key.mjs +108 -0
- package/src/session-store.mjs +249 -0
- package/src/start.mjs +502 -0
- package/src/subscriptions.mjs +208 -0
- package/src/triggers.mjs +491 -0
- package/src/up.mjs +315 -0
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
|
|
2
|
+
import { configError } from "./config.mjs";
|
|
3
|
+
import { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY } from "./exit-code.mjs";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* The job orchestration. Deliberately a pure-ish function over INJECTED side-effecting deps, so
|
|
7
|
+
* the money-safety ORDER can be tested without GitHub, Docker, or Redis.
|
|
8
|
+
*
|
|
9
|
+
* The order is the contract, and every step before `runContainer` must be free of provider spend:
|
|
10
|
+
*
|
|
11
|
+
* 0. refuse a job image this host does not have -- INT-CONTAINER-RUNTIME-CONTRACT
|
|
12
|
+
* 1. mint a scoped token (GitHub jobs, and local jobs opted in via `github: true`)
|
|
13
|
+
* -- CONST-TOKEN-SCOPED-PER-JOB
|
|
14
|
+
* 2. REFUSE an unprotected default branch (GitHub jobs only -- a local job has no repo)
|
|
15
|
+
* -- REQ-BRANCH-PROTECTION-PRECONDITION
|
|
16
|
+
* 3. resolve the default-branch SHA (fresh API), clone at it, materialise .pi/, write the prompt
|
|
17
|
+
* 4. reserve a budget slot -- CONST-BUDGET-BEFORE-TOKENS
|
|
18
|
+
* 5. ONLY NOW run the container (the only step that spends provider tokens)
|
|
19
|
+
* 6. map the container exit code to retry-vs-success
|
|
20
|
+
*
|
|
21
|
+
* Budget is reserved as late as possible but strictly before the container, so a refusal from an
|
|
22
|
+
* earlier free gate (unprotected repo, clone failure) never consumes a daily slot. The container
|
|
23
|
+
* is the only thing that spends money, so "before tokens" means "before this line".
|
|
24
|
+
*
|
|
25
|
+
* Returns a result object on a non-retryable outcome; THROWS on a retryable (infra) one so BullMQ
|
|
26
|
+
* retries per `attempts`. The caller (the BullMQ processor) turns the thrown/returned distinction
|
|
27
|
+
* into the queue's retry behaviour -- that is INT-RUNNER-EXIT-CODE-PROTOCOL.
|
|
28
|
+
*/
|
|
29
|
+
export async function runJob(job, deps) {
|
|
30
|
+
const {
|
|
31
|
+
redis,
|
|
32
|
+
caps, // { day, week, month }; week/month null when that window is disabled (REQ-SPEND-CAPS-MULTI-WINDOW)
|
|
33
|
+
softHoldPct, // int 1-99 or null; the soft-hold band applied to every active window
|
|
34
|
+
tokenCap = null, // int or null; the daily TOKEN cap (issue #25). Check-AFTER, so it gates the NEXT job on prior spend
|
|
35
|
+
recordSpend = recordTokenSpend, // injected so the post-container INCRBY is testable/stubbable
|
|
36
|
+
// (job) => { ok } | { missing: <ref> } | { unavailable: <ref> }. The pre-spend check that the image
|
|
37
|
+
// this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
|
|
38
|
+
// omits it behaves exactly as before -- the container's own failure stays the backstop.
|
|
39
|
+
imagePreflight = async () => ({ ok: true }),
|
|
40
|
+
// (session, { piVersion }) => { promoted, reason, bytes }. Promotes this job's transcript back into
|
|
41
|
+
// the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
|
|
42
|
+
// it behaves exactly as before -- no store, no promotion, no session in the record.
|
|
43
|
+
promoteSession = () => null,
|
|
44
|
+
// (job) => scoped short-lived token. Takes the JOB, not the repo: which forge mints -- and therefore
|
|
45
|
+
// which credential the container gets -- is a property of `job.kind`, and only the wiring knows the
|
|
46
|
+
// map. Called for forge-backed jobs and for local jobs opted in via `github: true`; unflagged local
|
|
47
|
+
// jobs never mint (token stays null).
|
|
48
|
+
mintToken,
|
|
49
|
+
isDefaultBranchProtected, // (job, token) => boolean; same reason -- the forge is the job's, not the process's
|
|
50
|
+
prepareWorkspace, // (job, token) => { workspaceDir, jobDir } (clone+materialise+prompt)
|
|
51
|
+
// runContainer({ job, token, prepared, name, signal }) => { code, aborted, turns, tokens, session, usage }. It MUST honour
|
|
52
|
+
// `signal`: stop the container on abort, and reject/exit promptly if `signal.aborted` is already
|
|
53
|
+
// true at entry (the timeout can fire during a slow prepare). The wiring injects name + signal.
|
|
54
|
+
runContainer,
|
|
55
|
+
cleanup, // (dirs) => void
|
|
56
|
+
comment, // (job, text) => void (issue status; no-op for local jobs)
|
|
57
|
+
log = () => {},
|
|
58
|
+
// The outbox chain collector (INT-OUTBOX-CONTRACT). No-op default so a job whose wiring omits it --
|
|
59
|
+
// or a github job with no /outbox -- chains nothing. It NEVER throws (outbox.mjs), so its counts are
|
|
60
|
+
// additive telemetry that can never flip the parent's completed outcome (CONST-RETRY-INFRA-ONLY).
|
|
61
|
+
collectChain = async () => ({ enqueued: 0, refused: 0 }),
|
|
62
|
+
now = new Date(),
|
|
63
|
+
} = deps;
|
|
64
|
+
|
|
65
|
+
// "Forge-backed" is the negation of local, not an enumeration of forges: a job that is not editing a
|
|
66
|
+
// folder on this host is working against a remote, and every gate below applies for the same reason
|
|
67
|
+
// regardless of WHICH remote. Written this way so a new forge inherits the gates rather than having to
|
|
68
|
+
// be added to them -- the failure mode of an enumeration is a forge that silently skips a money gate.
|
|
69
|
+
const isForgeBacked = job.kind !== "local";
|
|
70
|
+
// A local job opted in via `github: true` (cron trigger opt-in, INT-TRIGGERS-FILE-CONTRACT) mints the
|
|
71
|
+
// same scoped per-job token the github path mints (CONST-TOKEN-SCOPED-PER-JOB). Unflagged local jobs
|
|
72
|
+
// stay tokenless, exactly as before.
|
|
73
|
+
const wantsForgeToken = isForgeBacked || job.github === true;
|
|
74
|
+
let token = null;
|
|
75
|
+
let prepared = null;
|
|
76
|
+
let reserved = false;
|
|
77
|
+
|
|
78
|
+
try {
|
|
79
|
+
// The job image must exist on THIS host before anything else happens. Free, determinate and
|
|
80
|
+
// credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
|
|
81
|
+
// image refuses without minting a credential it will not use, cloning a repo it will not read, or
|
|
82
|
+
// burning a cap slot. Jobs run with --pull=never (docker-run.mjs), so an absent image is never
|
|
83
|
+
// fetched -- this refusal IS the whole diagnosis, not a race with a background pull.
|
|
84
|
+
const img = await imagePreflight(job);
|
|
85
|
+
// The image's declared pi version, read on the inspect the preflight already ran. Needed BEFORE the
|
|
86
|
+
// container starts, because a transcript written by a different pi may hold tool-call arguments the
|
|
87
|
+
// current schema no longer accepts -- so the resume has to be refused, not repaired mid-run. Null
|
|
88
|
+
// when the image declares none, which downstream means "never resume": the safe direction.
|
|
89
|
+
const piVersion = img.piVersion ?? null;
|
|
90
|
+
if (img.missing) {
|
|
91
|
+
await comment(job, `Refused: the job image "${img.missing}" is not present on the worker host. Not run.`);
|
|
92
|
+
log("refused_image_missing", { image: img.missing });
|
|
93
|
+
// The image ref is operator-authored config (PI_JOB_IMAGE), never payload, so naming it is PII-safe
|
|
94
|
+
// -- the same class as `repo` above.
|
|
95
|
+
// exitCode/turns/tokens null and budgetReserved false: refused pre-container AND pre-reserve.
|
|
96
|
+
// provider/model ride every terminal result from here down (INT-RUN-HISTORY-FILE-CONTRACT):
|
|
97
|
+
// runJob's `job` IS the effectiveJob (index.mjs), so these are the HOST-effective,
|
|
98
|
+
// overlay-resolved dispatch facts -- never anything a container printed -- and even a
|
|
99
|
+
// pre-container refusal attributes which (provider, model) it was dispatched for. There is
|
|
100
|
+
// deliberately NO `usage` key on the pre-container branches: no run, no ledger, and
|
|
101
|
+
// buildRecord defaults the absent field to null.
|
|
102
|
+
return { outcome: "policy", reason: "job-image-missing", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
103
|
+
}
|
|
104
|
+
if (img.forgeUnsupported) {
|
|
105
|
+
// The image is present and says it cannot serve this forge -- it ships no CLI for it. Determinate,
|
|
106
|
+
// so a refusal rather than a retry, and pre-spend, because the alternative is a paid container
|
|
107
|
+
// that fails at step 3 on every single delivery with nothing to distinguish it from a bad run.
|
|
108
|
+
//
|
|
109
|
+
// The message names the LIKELY CAUSE rather than the label that detected it: a trigger that
|
|
110
|
+
// forgot `run.image`. The failure is upstream of the thing that noticed it, and an operator
|
|
111
|
+
// reading "the image does not declare azure" has further to walk than one reading "set run.image".
|
|
112
|
+
await comment(
|
|
113
|
+
job,
|
|
114
|
+
`Refused: the job image "${img.forgeUnsupported}" does not support ${img.kind} jobs (it declares: ${img.declared.join(", ")}). Set this trigger's \`run.image\` to an image that does. Not run.`,
|
|
115
|
+
);
|
|
116
|
+
log("refused_image_forge_unsupported", { image: img.forgeUnsupported, kind: img.kind, declared: img.declared });
|
|
117
|
+
return { outcome: "policy", reason: "job-image-forge-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
118
|
+
}
|
|
119
|
+
if (img.replicaUnsupported) {
|
|
120
|
+
// The image is present and does not declare replica support (REQ-REPLICA-RUNS), so its baked
|
|
121
|
+
// HARD_RULES.md predates the amendment and still hard-codes `pi/issue-<n>` as a SYSTEM rule --
|
|
122
|
+
// which the model treats as authoritative over the user prompt naming `pi/issue-<n>-r2`. Both
|
|
123
|
+
// replicas would push to one branch: not an error, just the push race the feature exists to
|
|
124
|
+
// avoid, with two runs billed and one pull request to show for it.
|
|
125
|
+
//
|
|
126
|
+
// Determinate, so a refusal rather than a retry, and pre-spend, because no version of this gets
|
|
127
|
+
// better by running. Like the forge branch above, the message names the FIX rather than the label
|
|
128
|
+
// that noticed it -- an operator reading "rebuild the image" is already where they need to be.
|
|
129
|
+
await comment(
|
|
130
|
+
job,
|
|
131
|
+
`Refused: the job image "${img.replicaUnsupported}" does not declare replica support (\`dev.pi-dispatch.capabilities\` ${img.declared.length > 0 ? `declares: ${img.declared.join(", ")}` : "is absent"}), so its baked guardrails would name the wrong branch. Rebuild the image from a version that has this feature. Not run.`,
|
|
132
|
+
);
|
|
133
|
+
log("refused_image_replicas_unsupported", { image: img.replicaUnsupported, declared: img.declared });
|
|
134
|
+
return { outcome: "policy", reason: "job-image-replicas-unsupported", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
135
|
+
}
|
|
136
|
+
if (img.unavailable) {
|
|
137
|
+
// docker itself did not answer -- transient infra, NOT a determinate refusal. THROWN so BullMQ
|
|
138
|
+
// retries (CONST-RETRY-INFRA-ONLY). `container-never-started` is literally true here, and it reuses
|
|
139
|
+
// the refund path below: a no-op pre-reserve, and still honest if this gate ever moves.
|
|
140
|
+
// provider/model attribute even this pre-container death; no usage -- nothing ran to emit one.
|
|
141
|
+
throw new InfraRetry("docker unavailable, image preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
if (wantsForgeToken) {
|
|
145
|
+
token = await mintToken(job);
|
|
146
|
+
|
|
147
|
+
// Defense-in-depth at the DI seam: mintToken is injected, so we cannot assume it routed
|
|
148
|
+
// through get-token's own empty-token guard. An empty credential here would reach
|
|
149
|
+
// env-allowlist's `if (githubToken)` as a falsy value -> GITHUB_TOKEN omitted -> an
|
|
150
|
+
// anonymous paid run. Refuse before reserveBudget so a bad token burns no cap slot.
|
|
151
|
+
if (typeof token !== "string" || token.trim() === "") {
|
|
152
|
+
throw configError("mintToken returned an empty credential");
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
if (isForgeBacked) {
|
|
157
|
+
// REQ-BRANCH-PROTECTION-PRECONDITION. The agent's token can merge, so branch protection is the
|
|
158
|
+
// only technical barrier to a self-merge. Refuse before spending anything. Forge-backed jobs
|
|
159
|
+
// only: a local job has no remote branch to protect.
|
|
160
|
+
if (!(await isDefaultBranchProtected(job, token))) {
|
|
161
|
+
await comment(job, "Refused: the default branch is not protected. See SECURITY.md.");
|
|
162
|
+
log("refused_unprotected", { repo: job.repo });
|
|
163
|
+
// exitCode/turns/tokens null: refused pre-container, so no container exit, turn, or token count exists.
|
|
164
|
+
return { outcome: "policy", reason: "unprotected-branch", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
165
|
+
}
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
prepared = await prepareWorkspace(job, token, { piVersion }); // resolves SHA, clones, materialises .pi/, writes prompt
|
|
169
|
+
|
|
170
|
+
// A determinate prepare refusal (e.g. sha-gone: the default branch advanced past the resolved
|
|
171
|
+
// tip) is POLICY -- return before reserveBudget so it burns no cap slot and is never retried.
|
|
172
|
+
// Mirrors the branch-protection policy return above. Spread-plus-attribution: the prepare
|
|
173
|
+
// result keeps its own reason and fields, and the host-effective provider/model land beside
|
|
174
|
+
// them exactly as on every other terminal result.
|
|
175
|
+
if (prepared?.outcome === "policy") {
|
|
176
|
+
return { ...prepared, provider: job.provider ?? null, model: job.model ?? null };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Daily TOKEN cap (issue #25): the deliberate check-AFTER control. Token cost is only known
|
|
180
|
+
// post-run, so this cannot check-and-increment before the spend the way the job-count cap does
|
|
181
|
+
// (CONST-BUDGET-BEFORE-TOKENS). It is a read-only GET of prior jobs' recorded spend -- it consumes
|
|
182
|
+
// nothing, so it precedes reserveBudget's INCR and a refusal here burns no job-count slot. It can
|
|
183
|
+
// only stop the NEXT job once the day's accumulated spend has reached the cap; the actual INCRBY
|
|
184
|
+
// happens post-container via recordSpend. Reported before the job-count cap only because both are
|
|
185
|
+
// spend gates; the more-actionable branch-protection precondition is still reported first above --
|
|
186
|
+
// behind only the image check, which outranks it because a missing image blocks EVERY job of EVERY
|
|
187
|
+
// kind on this host, so it is the one the operator must fix first either way.
|
|
188
|
+
const tokenGate = await checkTokenCap(redis, { cap: tokenCap, now });
|
|
189
|
+
if (!tokenGate.allowed) {
|
|
190
|
+
await comment(job, `Over the daily token cap (${tokenGate.spent}/${tokenGate.cap} tokens). Not run.`);
|
|
191
|
+
log("over_token_budget", { spent: tokenGate.spent, cap: tokenGate.cap });
|
|
192
|
+
// budgetReserved false: refused before reserveBudget, so no job-count slot was consumed.
|
|
193
|
+
return { outcome: "policy", reason: "daily-token-cap", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
// Budget last-but-before-container. A refusal here spends nothing (no container starts). Reserves across
|
|
197
|
+
// every active window (day + optional week/month) and the soft-hold band in one atomic pass.
|
|
198
|
+
const budget = await reserveBudget(redis, { caps, softHoldPct, now });
|
|
199
|
+
reserved = true;
|
|
200
|
+
if (!budget.allowed) {
|
|
201
|
+
const w = budget.blockedWindow;
|
|
202
|
+
const win = budget.windows[w];
|
|
203
|
+
if (budget.reason === "soft-hold") {
|
|
204
|
+
await comment(job, `Soft-hold: ${w} spend ${win.reserved}/${win.cap} is inside the ${softHoldPct}% hold band. New starts paused; not run.`);
|
|
205
|
+
log("soft_hold", { window: w, reserved: win.reserved, cap: win.cap, pct: softHoldPct });
|
|
206
|
+
} else {
|
|
207
|
+
await comment(job, `Over the ${w} budget cap (${win.cap}). Not run.`);
|
|
208
|
+
log("over_budget", { window: w, reserved: win.reserved, cap: win.cap });
|
|
209
|
+
}
|
|
210
|
+
// budgetReserved true: the slot is reserved above and kept (a refused reservation still counts). Both
|
|
211
|
+
// over-budget and soft-hold are POLICY, RETURNED (not retried) -- the agent never ran.
|
|
212
|
+
return { outcome: "policy", reason: budget.reason, exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: true }; // return => not retried
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const { code, aborted, turns, tokens, session, usage } = await runContainer({ job, token, prepared });
|
|
216
|
+
log("container_exit", { exitCode: code, aborted });
|
|
217
|
+
|
|
218
|
+
// Record token spend post-run (the check-AFTER half of the lagging token cap). The container ran,
|
|
219
|
+
// so it spent real tokens on EVERY path that reaches here -- abort, completed, policy, AND the infra
|
|
220
|
+
// throws below (an exit-1 container still spent before failing). Record before classifying so all of
|
|
221
|
+
// them are accounted. Only when the cap is enabled (nothing reads the counter otherwise) and the run
|
|
222
|
+
// reported a positive total. NEVER throws: money is already spent, so a Redis blip here must not turn
|
|
223
|
+
// a completed paid job into a failure (mirrors the sink/comment/cleanup fault-isolation posture).
|
|
224
|
+
const tokensSpent = tokens?.total ?? 0;
|
|
225
|
+
if (tokenCap !== null && tokenCap !== undefined && tokensSpent > 0) {
|
|
226
|
+
await recordSpend(redis, tokensSpent, { now }).catch((err) => log("token_spend_error", { reason: err?.message }));
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// A WORKER-initiated stop (30-min timeout via cancelJob, or graceful-shutdown docker stop) kills
|
|
230
|
+
// the container -> exit 143/137. That is our decision, not an infra fault: it is POLICY and must
|
|
231
|
+
// NOT retry, or a wedged job re-runs into a second PR / double spend. Keyed on the abort FLAG,
|
|
232
|
+
// not the code -- an unbidden 137 (kernel OOM) carries `aborted: false`, falls to the switch, and
|
|
233
|
+
// stays infra-retryable.
|
|
234
|
+
// exitCode/turns/tokens carry the container's own exit, turn count, and usage totals; budgetReserved true post-reserve.
|
|
235
|
+
if (aborted) return { outcome: "policy", reason: "worker-abort", exitCode: code, turns, tokens, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
|
|
236
|
+
|
|
237
|
+
switch (code) {
|
|
238
|
+
case EXIT_COMPLETED: {
|
|
239
|
+
// The SOLE chain-collection point. Read the completed parent's /outbox and enqueue children
|
|
240
|
+
// BEFORE the `finally` deletes jobDir -- the await resolves inside this case, so the read
|
|
241
|
+
// finishes before control leaves to cleanup. NOT reached on any other branch (policy, abort,
|
|
242
|
+
// over-budget, infra): an InfraRetry job is retried, so chaining there would double-enqueue.
|
|
243
|
+
// collectChain never throws; chainEnqueued/chainRefused are additive telemetry only.
|
|
244
|
+
const chain = await collectChain({ job, prepared });
|
|
245
|
+
// COMPLETED-ONLY PROMOTION, and the exclusivity is the point rather than an optimisation.
|
|
246
|
+
// A policy or infra exit leaves the canonical transcript byte-identical to what it was
|
|
247
|
+
// before this run, so a retry starts from exactly what the first attempt did -- promote on
|
|
248
|
+
// every exit and "retry" quietly stops meaning re-run and starts meaning continue
|
|
249
|
+
// (CONST-RETRY-INFRA-ONLY). Same completed-only rule INT-OUTBOX-CONTRACT already uses, and
|
|
250
|
+
// it sits beside the chain collection for the same reason: both must happen before the
|
|
251
|
+
// `finally` deletes jobDir. Never throws.
|
|
252
|
+
const promoted = prepared.session ? promoteSession(prepared.session, { piVersion }) : null;
|
|
253
|
+
return {
|
|
254
|
+
outcome: "completed",
|
|
255
|
+
exitCode: code,
|
|
256
|
+
turns,
|
|
257
|
+
tokens,
|
|
258
|
+
// The validated per-model ledger the sink rebuilt off the exit line (parseExitUsage), or
|
|
259
|
+
// null for a fallback-metered or pre-ledger runner. `?? null` keeps the result shape
|
|
260
|
+
// stable under an injected runContainer that predates the field.
|
|
261
|
+
usage: usage ?? null,
|
|
262
|
+
provider: job.provider ?? null,
|
|
263
|
+
model: job.model ?? null,
|
|
264
|
+
session: mergeSession(prepared, session, promoted),
|
|
265
|
+
budgetReserved: true,
|
|
266
|
+
chainEnqueued: chain.enqueued,
|
|
267
|
+
chainRefused: chain.refused,
|
|
268
|
+
};
|
|
269
|
+
}
|
|
270
|
+
case EXIT_POLICY:
|
|
271
|
+
// A policy exit still ran a paid container, so it carries the ledger like the completed
|
|
272
|
+
// branch does -- the spend is real whichever way the runner classified itself.
|
|
273
|
+
return { outcome: "policy", reason: "runner-policy", exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
|
|
274
|
+
case EXIT_INFRA:
|
|
275
|
+
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
276
|
+
case 125: // `docker run` itself failed (unusable image reference, bad flag)
|
|
277
|
+
case 126: // the entrypoint exists but is not executable
|
|
278
|
+
case 127: // the entrypoint was not found
|
|
279
|
+
// In all three docker never handed control to the runner, so NOTHING was spent -- which is
|
|
280
|
+
// exactly what `container-never-started` means, and it reuses the refund below rather than
|
|
281
|
+
// keeping a slot the agent never used. These used to fall to `default:`, which kept the slot
|
|
282
|
+
// AND retried, burning a second one. The preflight above converts the KNOWABLE case (an absent
|
|
283
|
+
// image) into a pre-spend policy refusal; a 125 that survives it is a race (the image was
|
|
284
|
+
// removed between the inspect and the run) or a docker-side fault we did not foresee --
|
|
285
|
+
// genuinely infra, and now with a retry that costs nothing.
|
|
286
|
+
throw new InfraRetry(`docker could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
287
|
+
default:
|
|
288
|
+
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
289
|
+
}
|
|
290
|
+
} catch (e) {
|
|
291
|
+
// A spawn fault (docker daemon down / binary missing) reserved a slot but never started a
|
|
292
|
+
// container, so nothing was spent -- give the slot back before the retry. Every other throw
|
|
293
|
+
// here (exit-1 infra, unknown exit) means the container ran and legitimately spent its slot,
|
|
294
|
+
// so `reason` gates the release to the never-started case only. Guarded on `reserved` and run
|
|
295
|
+
// once per invocation; a BullMQ retry reserves afresh, so this cannot double-release.
|
|
296
|
+
// budgetReserved reflects whether a slot stays spent: false when never-started refunds below,
|
|
297
|
+
// true for a real container that ran and spent (exit-1 infra / unknown exit).
|
|
298
|
+
if (e instanceof InfraRetry) e.budgetReserved = reserved && e.reason !== "container-never-started";
|
|
299
|
+
if (reserved && e instanceof InfraRetry && e.reason === "container-never-started") {
|
|
300
|
+
await releaseBudget(redis, { caps, now });
|
|
301
|
+
}
|
|
302
|
+
throw e;
|
|
303
|
+
} finally {
|
|
304
|
+
if (prepared) await cleanup(prepared).catch(() => {});
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
/** Thrown for the retryable (infra) class only. The BullMQ processor lets this propagate to retry. */
|
|
309
|
+
/**
|
|
310
|
+
* The one `session` object the run record carries, from the host's intent and the container's report
|
|
311
|
+
* (INT-RUN-HISTORY-FILE-CONTRACT).
|
|
312
|
+
*
|
|
313
|
+
* Both halves matter and neither is sufficient. The host knows whether a key resolved and which gate
|
|
314
|
+
* refused; only the container knows what pi actually did with the file it was handed. A host that staged
|
|
315
|
+
* a transcript while the runner reports `resumed: false` is a real event -- a corrupt file, a degrade --
|
|
316
|
+
* and with one number alone it is indistinguishable from an ordinary cold start.
|
|
317
|
+
*
|
|
318
|
+
* The runner's verdict WINS on `resumed`, because it is the one that observed the outcome. The host's
|
|
319
|
+
* reason is kept when the runner has none to give (a container that died before its exit line).
|
|
320
|
+
*
|
|
321
|
+
* PII-free by construction: a boolean, a fixed enum, an integer. The key and the branch name are
|
|
322
|
+
* deliberately absent -- this record holds no attacker-chosen string, and a branch name is one.
|
|
323
|
+
*/
|
|
324
|
+
function mergeSession(prepared, fromRunner, promoted = null) {
|
|
325
|
+
const host = prepared?.session;
|
|
326
|
+
if (!host && !fromRunner) return null;
|
|
327
|
+
return {
|
|
328
|
+
resumed: fromRunner ? fromRunner.resumed : false,
|
|
329
|
+
// A promotion that was refused is the more useful reason to surface: "locked" or
|
|
330
|
+
// "not-a-regular-file" says why the NEXT run will cold-start, which is the thing an operator
|
|
331
|
+
// chasing "it never resumes" needs. It only ever replaces a reason on the completed path.
|
|
332
|
+
reason: (promoted && !promoted.promoted ? promoted.reason : null) ?? fromRunner?.reason ?? host?.reason ?? null,
|
|
333
|
+
bytes: promoted?.bytes ?? host?.bytes ?? null,
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
export class InfraRetry extends Error {
|
|
338
|
+
constructor(message, { cause, reason, exitCode, turns, tokens, session, usage, provider, model, budgetReserved } = {}) {
|
|
339
|
+
super(message, cause ? { cause } : undefined);
|
|
340
|
+
this.name = "InfraRetry";
|
|
341
|
+
this.piDispatchRetry = true;
|
|
342
|
+
this.reason = reason ?? message;
|
|
343
|
+
this.exitCode = exitCode ?? null;
|
|
344
|
+
this.turns = turns ?? null;
|
|
345
|
+
this.tokens = tokens ?? null;
|
|
346
|
+
// A deliberate in-passing repair: the EXIT_INFRA throw has passed `session` since the resume
|
|
347
|
+
// feature landed, but this destructure never read it, so every infra-retry record silently
|
|
348
|
+
// recorded session:null and a degrade seen only on a retried attempt left no trace. Latent
|
|
349
|
+
// because buildRecord's `?? null` made the drop indistinguishable from an honest absence.
|
|
350
|
+
this.session = session ?? null;
|
|
351
|
+
// The usage-ledger trio (INT-RUN-HISTORY-FILE-CONTRACT): carried on the throw path so a
|
|
352
|
+
// catch-path record attributes exactly what the return path would have.
|
|
353
|
+
this.usage = usage ?? null;
|
|
354
|
+
this.provider = provider ?? null;
|
|
355
|
+
this.model = model ?? null;
|
|
356
|
+
this.budgetReserved = budgetReserved ?? null;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
export { EXIT_COMPLETED, EXIT_INFRA, EXIT_POLICY };
|
package/src/queue.mjs
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import { Queue } from "bullmq";
|
|
2
|
+
import { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId } from "./job-id.mjs";
|
|
3
|
+
import { targetSeparator } from "./forges.mjs";
|
|
4
|
+
|
|
5
|
+
export const QUEUE = "pi-jobs";
|
|
6
|
+
export { chainedJobId, localJobId, deliveryJobId, gitlabDeliveryJobId, forgeDeliveryJobId };
|
|
7
|
+
|
|
8
|
+
export function makeQueue(connection) {
|
|
9
|
+
return new Queue(QUEUE, { connection });
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Enqueue a local-folder job. Returns the jobId. The data shape is what the processor's runJob
|
|
14
|
+
* consumes (kind/folder/flow/task/provider/model/maxTurns).
|
|
15
|
+
*
|
|
16
|
+
* removeOnComplete keeps the dedup window ~= the retention. Unlike webhooks, local jobs are not
|
|
17
|
+
* redelivered, so a modest window is enough.
|
|
18
|
+
*/
|
|
19
|
+
export async function enqueueLocalJob(queue, { folder, flow, task, provider, model, maxTurns, image, chainDepth, parentJobId, jobId, now = new Date() }) {
|
|
20
|
+
const minute = now.toISOString().slice(0, 16); // YYYY-MM-DDTHH:MM -- the dedup window
|
|
21
|
+
// A caller-supplied jobId (the outbox collector's retry-idempotent chainedJobId) wins; otherwise the
|
|
22
|
+
// minute-windowed localJobId is the dedup key.
|
|
23
|
+
const id = jobId ?? localJobId({ folder, flow, task, minute });
|
|
24
|
+
// image/chainDepth/parentJobId land on `data` only when present, so a plain non-chained job's data is
|
|
25
|
+
// byte-identical. `image` is the container image this job runs in (INT-TRIGGERS-FILE-CONTRACT); absent
|
|
26
|
+
// resolves the deployment default at job start, never a value frozen here.
|
|
27
|
+
const data = {
|
|
28
|
+
kind: "local",
|
|
29
|
+
folder,
|
|
30
|
+
flow,
|
|
31
|
+
task,
|
|
32
|
+
provider,
|
|
33
|
+
model,
|
|
34
|
+
maxTurns,
|
|
35
|
+
...(image !== undefined && { image }),
|
|
36
|
+
...(chainDepth !== undefined && { chainDepth }),
|
|
37
|
+
...(parentJobId !== undefined && { parentJobId }),
|
|
38
|
+
};
|
|
39
|
+
await queue.add("local", data, {
|
|
40
|
+
jobId: id,
|
|
41
|
+
attempts: 2,
|
|
42
|
+
backoff: { type: "exponential", delay: 60_000 },
|
|
43
|
+
removeOnComplete: { age: 24 * 3600 },
|
|
44
|
+
removeOnFail: { age: 7 * 24 * 3600 },
|
|
45
|
+
});
|
|
46
|
+
return id;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// Coalesces rapid re-label spam; the GUID jobId + 31d retention handle exact redelivery, so this
|
|
50
|
+
// window only needs to absorb burst re-labels, not the full redelivery window.
|
|
51
|
+
const SEMANTIC_WINDOW_MS = 10 * 60 * 1000;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Enqueue a GitHub-triggered job. Returns the jobId. The data shape is what prepare/runJob consumes
|
|
55
|
+
* for the github kind. No `sha` field: the commit is resolved fresh in prepare (C1), so baking a
|
|
56
|
+
* possibly-stale sha here would only race the branch head.
|
|
57
|
+
*
|
|
58
|
+
* `target` is the discriminated subject of the job -- `{ type:"issue"|"pull_request", number, title,
|
|
59
|
+
* body, ... }` -- built by the receiver's filter from the INT-WEBHOOK-PAYLOAD-SUBSET fields. Its `number`
|
|
60
|
+
* keys the semantic dedup window; GitHub issues and PRs share one per-repo number sequence, so the key is
|
|
61
|
+
* collision-free without encoding the type. That is a fact about GitHub, not about forges -- see
|
|
62
|
+
* `enqueueGitLabJob`, where they are separate sequences and the type has to be in the key.
|
|
63
|
+
*
|
|
64
|
+
* Two dedup layers, ADDITIVE and independent:
|
|
65
|
+
* - `jobId` (the delivery GUID) is exact-per-delivery: a redelivered webhook resolves to the same
|
|
66
|
+
* id and BullMQ's `EXISTS jobId` rejects it -- REQ-DEDUP-BY-DELIVERY-GUID.
|
|
67
|
+
* - `deduplication` keys on `repo#number:flow` for SEMANTIC_WINDOW_MS: distinct GUIDs from rapid
|
|
68
|
+
* re-labels or repeated PR pushes coalesce to one active job. It coexists with jobId; it does not
|
|
69
|
+
* replace it.
|
|
70
|
+
*/
|
|
71
|
+
export async function enqueueGitHubJob(queue, fields) {
|
|
72
|
+
return await enqueueForgeJob(queue, "github", fields);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Enqueue a GitLab-triggered job. Structurally the twin of `enqueueGitHubJob` -- and now literally the
|
|
77
|
+
* same body, because everything that differs between them turned out to be two table entries.
|
|
78
|
+
*
|
|
79
|
+
* `projectId` rides the data because every GitLab API path the worker needs takes the numeric project id.
|
|
80
|
+
* A GitLab project path is `group/subgroup/project` with no fixed segment count, so the `owner/name` split
|
|
81
|
+
* the GitHub path uses does not merely fail on one, it SUCCEEDS wrongly: both halves come back non-empty
|
|
82
|
+
* and the project silently becomes its own parent group. Carrying the id sidesteps the grammar entirely;
|
|
83
|
+
* `repo` stays as the human-readable label for logs, run history and pause-window scopes.
|
|
84
|
+
*/
|
|
85
|
+
export async function enqueueGitLabJob(queue, fields) {
|
|
86
|
+
return await enqueueForgeJob(queue, "gitlab", fields);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Enqueue a forge-triggered job of any kind. Returns the jobId.
|
|
91
|
+
*
|
|
92
|
+
* The two named wrappers above are spellings of this. They were separate bodies until a third and fourth
|
|
93
|
+
* forge made that four copies of the retention window, the retry policy, the backoff and BOTH dedup
|
|
94
|
+
* layers -- four places for one of them to be quietly weakened while every test stayed green.
|
|
95
|
+
*
|
|
96
|
+
* The semantic dedup key encodes the TARGET TYPE through `targetSeparator`, which GitHub alone does not
|
|
97
|
+
* need: it numbers issues and pull requests from one per-repo sequence, so `repo#7` names exactly one
|
|
98
|
+
* thing. GitLab numbers them separately, so issue #5 and merge request !5 would collide on `project#5:flow`
|
|
99
|
+
* -- one silently coalescing into the other's 10-minute window and never running. The separator is each
|
|
100
|
+
* forge's own notation, and it lives in the table because it is a fact about the forge.
|
|
101
|
+
*
|
|
102
|
+
* Forge-specific data fields are listed EXPLICITLY rather than collected with a rest spread. A spread
|
|
103
|
+
* would persist whatever a caller happened to pass into durable job data, and this object is copied
|
|
104
|
+
* verbatim into `/job/event.json` -- a place where an unreviewed field has no business.
|
|
105
|
+
*
|
|
106
|
+
* REPLICAS (REQ-REPLICA-RUNS) are the one case where one delivery becomes more than one job, and BOTH dedup
|
|
107
|
+
* layers have to be told, not just the id. The caller loops and passes `replica` 1..N; each pass is an
|
|
108
|
+
* ordinary enqueue with a distinct id and a distinct semantic key. The `replica` suffix on the dedup id is
|
|
109
|
+
* added ONLY when a replica is set, so re-deliveries of each replica still coalesce within the 10-minute
|
|
110
|
+
* window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
|
|
111
|
+
* has always been.
|
|
112
|
+
*/
|
|
113
|
+
export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, trigger, provider, model, maxTurns, packages, image, resume, replica, replicas }) {
|
|
114
|
+
const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
|
|
115
|
+
// `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
|
|
116
|
+
// come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
|
|
117
|
+
// only when the filter resolved one, exactly like chainDepth/parentJobId above, so an unflagged trigger's
|
|
118
|
+
// job data is byte-identical. Both sit at JOB level, never inside `trigger` -- that object is descriptive
|
|
119
|
+
// and is copied verbatim into /job/event.json, where an execution knob has no business.
|
|
120
|
+
const data = {
|
|
121
|
+
kind,
|
|
122
|
+
repo,
|
|
123
|
+
...(projectId !== undefined && { projectId }),
|
|
124
|
+
// Azure's org/project/repository triple, alongside the human-readable `repo` -- the same split gitlab
|
|
125
|
+
// makes with `projectId`, and for the same reason: every Azure API path takes ids and names this label
|
|
126
|
+
// cannot be reassembled into without guessing.
|
|
127
|
+
...(azure !== undefined && { azure }),
|
|
128
|
+
target,
|
|
129
|
+
flow,
|
|
130
|
+
trigger,
|
|
131
|
+
provider,
|
|
132
|
+
model,
|
|
133
|
+
maxTurns,
|
|
134
|
+
...(packages !== undefined && { packages }),
|
|
135
|
+
...(image !== undefined && { image }),
|
|
136
|
+
...(resume !== undefined && { resume }),
|
|
137
|
+
// Conditional for the same reason packages/image/resume are: an unflagged job's data must keep
|
|
138
|
+
// exactly the keys it has today. `replica` is this job's 1-based index and `replicas` the set size;
|
|
139
|
+
// both are integers, so the run record they land in stays PII-free by construction.
|
|
140
|
+
...(replica !== undefined && { replica }),
|
|
141
|
+
...(replicas !== undefined && { replicas }),
|
|
142
|
+
};
|
|
143
|
+
await queue.add(kind, data, {
|
|
144
|
+
jobId,
|
|
145
|
+
deduplication: { id: `${repo}${targetSeparator(kind, target?.type)}${target.number}:${flow}${replica !== undefined ? `:r${replica}` : ""}`, ttl: SEMANTIC_WINDOW_MS }, // ttl in ms
|
|
146
|
+
attempts: 2,
|
|
147
|
+
backoff: { type: "exponential", delay: 60_000 },
|
|
148
|
+
removeOnComplete: { age: 31 * 24 * 3600 }, // age in seconds -- do not cross units with the ms ttl above
|
|
149
|
+
removeOnFail: { age: 31 * 24 * 3600 },
|
|
150
|
+
});
|
|
151
|
+
return jobId;
|
|
152
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
2
|
+
import { buildDockerRunArgs, CONTAINER_SESSION_FILE } from "./docker-run.mjs";
|
|
3
|
+
import { buildContainerEnv } from "./env-allowlist.mjs";
|
|
4
|
+
import { resolveJobImage } from "./image-preflight.mjs";
|
|
5
|
+
import { InfraRetry } from "./processor.mjs";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* The real `runContainer` the processor injects. Launches one job container and returns
|
|
9
|
+
* `{ code, aborted, turns, tokens, session, usage }`, where `aborted` records whether the WORKER initiated the stop (docker stop on
|
|
10
|
+
* the 30-min timeout or graceful shutdown), which the processor classifies as POLICY (no retry) per
|
|
11
|
+
* INT-RUNNER-EXIT-CODE-PROTOCOL. The numeric `code` alone cannot say this: a worker SIGKILL and a
|
|
12
|
+
* kernel OOM both surface as 137, so the abort FLAG -- not the code -- is the discriminator.
|
|
13
|
+
*
|
|
14
|
+
* `spawn` (not execFile) because a non-zero exit is NORMAL here: exit 1 (infra) and 2 (policy) are
|
|
15
|
+
* expected outcomes, not errors to reject on. The exit code comes from the `close` event.
|
|
16
|
+
*
|
|
17
|
+
* The container is stopped on abort by the worker wiring (index.mjs onAbort -> docker stop), which
|
|
18
|
+
* causes `docker run` to exit and this promise to resolve. We only handle the entry case here: if
|
|
19
|
+
* the signal is ALREADY aborted (the 30-min timeout fired during a slow prepare), do not start a
|
|
20
|
+
* container at all.
|
|
21
|
+
*
|
|
22
|
+
* Output is streamed to `onOutput` (default: the worker's stdout) so the operator watches the agent
|
|
23
|
+
* work on their own machine -- the natural local UX. When raw capture is enabled
|
|
24
|
+
* (`PI_CAPTURE_JOB_LOGS`), the same output is tee'd to a host-only, gitignored `logs/<jobId>.log`
|
|
25
|
+
* that is never mounted into the container and may contain agent-echoed issue text (PII). The
|
|
26
|
+
* worker's event log and the `.json` status record stay id-only.
|
|
27
|
+
*/
|
|
28
|
+
export function makeRunContainer({
|
|
29
|
+
image, // the DEPLOYMENT default (PI_JOB_IMAGE); a trigger's own run.image overrides it per job
|
|
30
|
+
hostEnv = process.env,
|
|
31
|
+
onOutput = (c) => process.stdout.write(c),
|
|
32
|
+
openJobLog = () => ({ write() {}, close: async () => ({ turns: null, tokens: null, session: null, usage: null }) }),
|
|
33
|
+
spawnFn = spawn,
|
|
34
|
+
globalPiDir = null, // REQ-GLOBAL-PI-OVERLAY: operator's global pi overlay dir, mounted :ro; null = off
|
|
35
|
+
allowGlobalExtensions = true, // REQ-GLOBAL-PI-OVERLAY: the staged overlay's extensions load unless PI_GLOBAL_ALLOW_EXTENSIONS=0
|
|
36
|
+
packagePaths = [], // REQ-GLOBAL-PI-OVERLAY: container paths of the operator-staged packages, resolved once at boot
|
|
37
|
+
forwardEnv = [],
|
|
38
|
+
authFromPi = false, // fall back to ~/.pi/agent/auth.json for the provider key when the env has none
|
|
39
|
+
forgeHosts = {}, // per-forge self-hosted instance URLs, so a forge CLI in the container talks to the right one
|
|
40
|
+
}) {
|
|
41
|
+
// async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
|
|
42
|
+
// a rejection, uniformly awaitable by the processor and by tests.
|
|
43
|
+
return async function runContainer({ job, token, prepared, name, signal }) {
|
|
44
|
+
if (signal?.aborted) return { code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null }; // killed before it could start
|
|
45
|
+
|
|
46
|
+
// Closed env allowlist: only the provider key + the declared PI_* vars. Throws (config) if
|
|
47
|
+
// the provider is unconfigured -- the processor turns that into a pre-spend refusal.
|
|
48
|
+
const env = buildContainerEnv({
|
|
49
|
+
provider: job.provider,
|
|
50
|
+
model: job.model,
|
|
51
|
+
maxTurns: job.maxTurns,
|
|
52
|
+
maxTokens: job.maxTokens, // optional per-job token budget (issue #25); undefined => runner meter only
|
|
53
|
+
jobId: name,
|
|
54
|
+
githubToken: token ?? undefined,
|
|
55
|
+
// Which forge minted it, so the token lands in that forge's own variable names and no other.
|
|
56
|
+
forgeKind: job?.kind,
|
|
57
|
+
forgeHosts,
|
|
58
|
+
hostEnv,
|
|
59
|
+
allowGlobalExtensions, // REQ-GLOBAL-PI-OVERLAY: false emits the explicit PI_GLOBAL_ALLOW_EXTENSIONS=0 opt-out
|
|
60
|
+
// REQ-GLOBAL-PI-OVERLAY: the per-job value comes off `job` (like maxTurns), the staged set off
|
|
61
|
+
// the closure (like allowGlobalExtensions) -- so a trigger can withhold what the operator staged.
|
|
62
|
+
// `!== false`, because staged packages LOAD unless a trigger explicitly opts out
|
|
63
|
+
// (INT-TRIGGERS-FILE-CONTRACT). The strictness that used to live in this `=== true` did not
|
|
64
|
+
// disappear, it moved: parseTriggers refuses any non-boolean run.packages fail-loud at load, so a
|
|
65
|
+
// hand-edited string "false" never becomes job data this comparison could misread as an opt-out.
|
|
66
|
+
packagePaths: job.packages === false ? [] : packagePaths,
|
|
67
|
+
forwardEnv, // extra host var names to forward (e.g. a custom provider's key)
|
|
68
|
+
// REQ-RESUMABLE-SESSION: the fixed container path, emitted only when this job HAS a transcript.
|
|
69
|
+
// The constant is imported rather than re-typed so the mount below and this variable name one
|
|
70
|
+
// path -- two literals is how they drift with both suites green.
|
|
71
|
+
sessionFile: prepared.session ? CONTAINER_SESSION_FILE : undefined,
|
|
72
|
+
authFromPi, // source the provider key from pi's auth.json when the env has none
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
const args = buildDockerRunArgs({
|
|
76
|
+
// Same split as packagePaths above: the per-job value off `job`, the deployment value off the closure,
|
|
77
|
+
// so a trigger can name its own toolchain (INT-TRIGGERS-FILE-CONTRACT). Resolved through the SAME
|
|
78
|
+
// function the pre-spend preflight uses (image-preflight.mjs), so the tag that was checked is the tag
|
|
79
|
+
// that runs -- one answer by construction, not two call sites that happen to agree.
|
|
80
|
+
image: resolveJobImage(job, image),
|
|
81
|
+
env,
|
|
82
|
+
jobDir: prepared.jobDir,
|
|
83
|
+
workspace: prepared.workspace,
|
|
84
|
+
outboxDir: prepared.outboxDir, // undefined for github jobs -> docker-run's guard skips the /outbox mount
|
|
85
|
+
// The job's OWN copy, under jobDir -- never the shared store. Undefined when the trigger did not
|
|
86
|
+
// arm run.resume or no key resolved, and docker-run's guard then skips the mount entirely.
|
|
87
|
+
sessionDir: prepared.session?.hostDir,
|
|
88
|
+
globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
|
|
89
|
+
name,
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
// Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
|
|
93
|
+
// sanitizes internally. No container mount, no env var -- the sink lives on this side only.
|
|
94
|
+
const sink = openJobLog(name);
|
|
95
|
+
|
|
96
|
+
return await new Promise((resolve, reject) => {
|
|
97
|
+
const child = spawnFn("docker", args, { stdio: ["ignore", "pipe", "pipe"] });
|
|
98
|
+
// A throwing sink.write is swallowed so a misbehaving sink cannot break the tee or hang the run.
|
|
99
|
+
const tee = (chunk) => {
|
|
100
|
+
onOutput(chunk);
|
|
101
|
+
try {
|
|
102
|
+
sink.write(chunk);
|
|
103
|
+
} catch {}
|
|
104
|
+
};
|
|
105
|
+
child.stdout?.on("data", tee);
|
|
106
|
+
child.stderr?.on("data", tee);
|
|
107
|
+
// docker not found / daemon down -- a transient infra fault, so tag it retryable
|
|
108
|
+
// (CONST-RETRY-INFRA-ONLY). `reason` also cues the processor to release the budget slot,
|
|
109
|
+
// since a container that never started spent nothing.
|
|
110
|
+
child.on("error", (err) => {
|
|
111
|
+
sink.close().catch(() => {}); // best-effort teardown; a rejecting close cannot leak an unhandled rejection
|
|
112
|
+
reject(new InfraRetry("container-never-started", { cause: err, reason: "container-never-started" }));
|
|
113
|
+
});
|
|
114
|
+
child.on("close", async (code) => {
|
|
115
|
+
const aborted = signal?.aborted === true; // capture BEFORE the await
|
|
116
|
+
// A rejecting sink.close is swallowed so a misbehaving sink cannot hang the run; turns/tokens/session/usage fall back to null.
|
|
117
|
+
let turns = null;
|
|
118
|
+
let tokens = null;
|
|
119
|
+
let session = null;
|
|
120
|
+
let usage = null;
|
|
121
|
+
try {
|
|
122
|
+
({ turns, tokens, session, usage } = await sink.close());
|
|
123
|
+
} catch {
|
|
124
|
+
turns = null;
|
|
125
|
+
tokens = null;
|
|
126
|
+
session = null;
|
|
127
|
+
usage = null;
|
|
128
|
+
}
|
|
129
|
+
resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage } : { code: code ?? 1, aborted: false, turns, tokens, session, usage });
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
};
|
|
133
|
+
}
|