@edgehero/pi-dispatch 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +15 -0
- package/deploy/docker-compose.yml +59 -0
- package/deploy/egress-proxy.conf +36 -0
- package/deploy/receiver.service +11 -0
- package/deploy/worker-env-wrapper.cmd +33 -4
- package/deploy/worker-env-wrapper.sh +38 -2
- package/package.json +1 -1
- package/src/azure-prompt.mjs +57 -9
- package/src/config.mjs +117 -6
- package/src/docker-run.mjs +16 -1
- package/src/doctor.mjs +368 -12
- package/src/egress.mjs +221 -0
- package/src/env-allowlist.mjs +16 -1
- package/src/forgejo-prompt.mjs +65 -11
- package/src/get-token.mjs +5 -3
- package/src/github-prompt.mjs +11 -2
- package/src/gitlab-prompt.mjs +65 -11
- package/src/init.mjs +30 -0
- package/src/packages.mjs +4 -1
- package/src/prepare-github.mjs +6 -4
- package/src/processor.mjs +43 -0
- package/src/run-container.mjs +32 -1
- package/src/run-history.mjs +1 -1
- package/src/sandbox-cli.mjs +24 -3
- package/src/sandbox.mjs +14 -1
- package/src/service.mjs +289 -25
- package/src/start.mjs +26 -0
- package/src/triggers.mjs +11 -12
- package/src/up.mjs +58 -0
package/src/processor.mjs
CHANGED
|
@@ -41,6 +41,9 @@ export async function runJob(job, deps) {
|
|
|
41
41
|
// this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
|
|
42
42
|
// omits it behaves exactly as before -- the container's own failure stays the backstop.
|
|
43
43
|
imagePreflight = async () => ({ ok: true }),
|
|
44
|
+
// REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
|
|
45
|
+
// deployment with no egress policy does -- which is also what the real factory returns when unarmed.
|
|
46
|
+
egressPreflight = async () => ({ ok: true }),
|
|
44
47
|
// (session, { piVersion }) => { promoted, reason, bytes }. Promotes this job's transcript back into
|
|
45
48
|
// the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
|
|
46
49
|
// it behaves exactly as before -- no store, no promotion, no session in the record.
|
|
@@ -184,6 +187,46 @@ export async function runJob(job, deps) {
|
|
|
184
187
|
throw new InfraRetry("docker unavailable, image preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
185
188
|
}
|
|
186
189
|
|
|
190
|
+
// REQ-EGRESS-ALLOWLIST. The egress policy this deployment claims must be able to serve this job
|
|
191
|
+
// BEFORE the job costs anything. It is one `docker inspect` when the policy is armed and ZERO spawns
|
|
192
|
+
// when it is not, so a deployment without one pays nothing at all.
|
|
193
|
+
//
|
|
194
|
+
// PLACEMENT, and it is the same ladder the image preflight sits at the top of. A missing proxy blocks
|
|
195
|
+
// EVERY job of EVERY kind on this host -- like a missing image -- and unlike a missing image it blocks
|
|
196
|
+
// them EXPENSIVELY: the container starts, the provider is unreachable, the runner exits 1, exit 1 is
|
|
197
|
+
// the retryable class, `attempts: 2`, and `releaseBudget` refunds only `container-never-started` --
|
|
198
|
+
// this container started. So each such job spends two job-count slots and buys nothing with either,
|
|
199
|
+
// and a cron-driven deployment empties its daily cap before anyone reads the first failure. That cost
|
|
200
|
+
// is what makes this a pre-spend gate rather than a doc: measured at three provider attempts,
|
|
201
|
+
// `Request timed out.`, exit 1, ~40 seconds, zero tokens (docs/egress.md).
|
|
202
|
+
//
|
|
203
|
+
// A RETURN, never a throw (CONST-RETRY-INFRA-ONLY): retrying never makes an absent proxy appear.
|
|
204
|
+
const egress = await egressPreflight(job);
|
|
205
|
+
if (egress.proxyMissing || egress.proxyStopped) {
|
|
206
|
+
const proxy = egress.proxyMissing ?? egress.proxyStopped;
|
|
207
|
+
const state = egress.proxyMissing ? "is not on this host" : "is not running";
|
|
208
|
+
await comment(job, `Refused: this deployment runs jobs behind an egress policy and its allowlist proxy "${proxy}" ${state}, so the job could not reach the provider and would burn its budget slot proving it. Start it with \`docker compose -f deploy/docker-compose.yml --profile egress up -d\`, or set PI_EGRESS=0 to run without an egress policy. Not run.`);
|
|
209
|
+
// The proxy's NAME is operator-authored deployment config, never payload -- the same PII class as
|
|
210
|
+
// the image ref on the refusal above.
|
|
211
|
+
log(egress.proxyMissing ? "refused_egress_proxy_missing" : "refused_egress_proxy_stopped", { proxy });
|
|
212
|
+
return {
|
|
213
|
+
outcome: "policy",
|
|
214
|
+
reason: egress.proxyMissing ? "egress-proxy-missing" : "egress-proxy-stopped",
|
|
215
|
+
exitCode: null,
|
|
216
|
+
turns: null,
|
|
217
|
+
tokens: null,
|
|
218
|
+
provider: job.provider ?? null,
|
|
219
|
+
model: job.model ?? null,
|
|
220
|
+
budgetReserved: false, // refused before reserveBudget, so no job-count slot was consumed
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
if (egress.unavailable) {
|
|
224
|
+
// The daemon did not answer, so this is indeterminate rather than a refusal -- the same
|
|
225
|
+
// determinate/indeterminate split the image preflight draws one gate up, and thrown for the same
|
|
226
|
+
// reason. Pre-reserve, so the refund below is a no-op and still honest if this gate ever moves.
|
|
227
|
+
throw new InfraRetry("docker unavailable, egress preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
|
|
228
|
+
}
|
|
229
|
+
|
|
187
230
|
// REQ-RESUMABLE-SESSION's one fail-CLOSED case. Everything else in that feature fails OPEN and
|
|
188
231
|
// NAMES itself -- absent, expired, too-large, unparseable, locked, promote-failed -- because a cold
|
|
189
232
|
// start is a correct run. This one cannot be: with no `sessionsDir`, resolveSession returns null
|
package/src/run-container.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import { buildDockerRunArgs, CONTAINER_SESSION_FILE } from "./docker-run.mjs";
|
|
3
|
+
import { createJobNetwork, networkNameFor, removeJobNetwork } from "./egress.mjs";
|
|
3
4
|
import { buildContainerEnv } from "./env-allowlist.mjs";
|
|
4
5
|
import { resolveJobImage } from "./image-preflight.mjs";
|
|
5
6
|
import { InfraRetry } from "./processor.mjs";
|
|
@@ -40,6 +41,8 @@ export function makeRunContainer({
|
|
|
40
41
|
forwardEnv = [],
|
|
41
42
|
authFromPi = false, // fall back to ~/.pi/agent/auth.json for the provider key when the env has none
|
|
42
43
|
forgeHosts = {}, // per-forge self-hosted instance URLs, so a forge CLI in the container talks to the right one
|
|
44
|
+
egress = false, // REQ-EGRESS-ALLOWLIST: put this job on its own --internal network behind the allowlist proxy
|
|
45
|
+
egressProxy, // the proxy component attached to that network; undefined = egress.mjs's default name
|
|
43
46
|
}) {
|
|
44
47
|
// async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
|
|
45
48
|
// a rejection, uniformly awaitable by the processor and by tests.
|
|
@@ -59,6 +62,8 @@ export function makeRunContainer({
|
|
|
59
62
|
forgeKind: job?.kind,
|
|
60
63
|
forgeHosts,
|
|
61
64
|
hostEnv,
|
|
65
|
+
egress, // REQ-EGRESS-ALLOWLIST: emits HTTPS_PROXY/HTTP_PROXY/NO_PROXY/NODE_USE_ENV_PROXY, or nothing
|
|
66
|
+
egressProxy,
|
|
62
67
|
allowGlobalExtensions, // REQ-GLOBAL-PI-OVERLAY: false emits the explicit PI_GLOBAL_ALLOW_EXTENSIONS=0 opt-out
|
|
63
68
|
// REQ-GLOBAL-PI-OVERLAY: the per-job value comes off `job` (like maxTurns), the staged set off
|
|
64
69
|
// the closure (like allowGlobalExtensions) -- so a trigger can withhold what the operator staged.
|
|
@@ -84,6 +89,10 @@ export function makeRunContainer({
|
|
|
84
89
|
authFromPi, // source the provider key from pi's auth.json when the env has none
|
|
85
90
|
});
|
|
86
91
|
|
|
92
|
+
// `-net` on this container's own name (egress.mjs). null when no policy is armed, and docker-run's
|
|
93
|
+
// guard then omits the flag entirely, so the argv is byte-identical to one built before this feature.
|
|
94
|
+
const network = egress ? networkNameFor(name) : null;
|
|
95
|
+
|
|
87
96
|
const args = buildDockerRunArgs({
|
|
88
97
|
// Same split as packagePaths above: the per-job value off `job`, the deployment value off the closure,
|
|
89
98
|
// so a trigger can name its own toolchain (INT-TRIGGERS-FILE-CONTRACT). Resolved through the SAME
|
|
@@ -99,13 +108,26 @@ export function makeRunContainer({
|
|
|
99
108
|
sessionDir: prepared.session?.hostDir,
|
|
100
109
|
globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
|
|
101
110
|
name,
|
|
111
|
+
network, // REQ-EGRESS-ALLOWLIST: null when no policy is armed, and the flag is then absent
|
|
102
112
|
});
|
|
103
113
|
|
|
114
|
+
// REQ-EGRESS-ALLOWLIST. This job's own --internal network, created here rather than at boot because
|
|
115
|
+
// it holds exactly two endpoints -- this container and the proxy -- and that is what makes job-to-job
|
|
116
|
+
// traffic structurally impossible rather than merely discouraged. A shared network could not do it:
|
|
117
|
+
// `enable_icc=false` would block job-to-job AND job-to-proxy, since ICC governs every container pair
|
|
118
|
+
// on the bridge and the proxy is a container.
|
|
119
|
+
//
|
|
120
|
+
// A failure to build it is INFRA, not policy: nothing has been spent, a retry may well succeed, and
|
|
121
|
+
// `container-never-started` is literally true, so the reservation is given back (processor.mjs).
|
|
122
|
+
if (network && !(await createJobNetwork(spawnFn, { network, proxy: egressProxy }))) {
|
|
123
|
+
throw new InfraRetry("container-never-started", { reason: "container-never-started" });
|
|
124
|
+
}
|
|
125
|
+
|
|
104
126
|
// Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
|
|
105
127
|
// sanitizes internally. No container mount, no env var -- the sink lives on this side only.
|
|
106
128
|
const sink = openJobLog(name);
|
|
107
129
|
|
|
108
|
-
|
|
130
|
+
const run = new Promise((resolve, reject) => {
|
|
109
131
|
const child = spawnFn("docker", args, { stdio: ["ignore", "pipe", "pipe"] });
|
|
110
132
|
// A throwing sink.write is swallowed so a misbehaving sink cannot break the tee or hang the run.
|
|
111
133
|
const tee = (chunk) => {
|
|
@@ -141,5 +163,14 @@ export function makeRunContainer({
|
|
|
141
163
|
resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage } : { code: code ?? 1, aborted: false, turns, tokens, session, usage });
|
|
142
164
|
});
|
|
143
165
|
});
|
|
166
|
+
|
|
167
|
+
// The network outlives the container by exactly this `finally`. Best-effort and never throwing: the
|
|
168
|
+
// container has already exited, its code is the job's answer, and a teardown fault must not rewrite
|
|
169
|
+
// that answer. What a failure leaves behind is a memberless network, which the boot reaper sweeps.
|
|
170
|
+
try {
|
|
171
|
+
return await run;
|
|
172
|
+
} finally {
|
|
173
|
+
if (network) await removeJobNetwork(spawnFn, { network, proxy: egressProxy });
|
|
174
|
+
}
|
|
144
175
|
};
|
|
145
176
|
}
|
package/src/run-history.mjs
CHANGED
|
@@ -246,7 +246,7 @@ function rebuildUsage(u) {
|
|
|
246
246
|
* path embeds the operator's OS account name.
|
|
247
247
|
*
|
|
248
248
|
* `reason` is a fixed enum passthrough (worker-abort | over-budget | unprotected-branch |
|
|
249
|
-
* runner-policy | job-image-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
|
|
249
|
+
* runner-policy | job-image-missing | egress-proxy-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
|
|
250
250
|
* default to `null` when the outcome does not carry them, so the record shape is stable whether or not
|
|
251
251
|
* the source reports those fields.
|
|
252
252
|
*/
|
package/src/sandbox-cli.mjs
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import { spawn } from "node:child_process";
|
|
1
2
|
import { parseArgs } from "node:util";
|
|
2
3
|
import { loadConfig } from "./config.mjs";
|
|
3
4
|
import { sanitizeJobId } from "./run-history.mjs";
|
|
5
|
+
import { createJobNetwork, egressEnv, networkNameFor, removeJobNetwork } from "./egress.mjs";
|
|
4
6
|
import { buildSandboxRunArgs, launchSandbox, listRunningSandboxes, parsePublish, resolveSandbox, sandboxContainerName } from "./sandbox.mjs";
|
|
5
7
|
import { listSandboxes, pinSandbox } from "./sandbox-store.mjs";
|
|
6
8
|
|
|
@@ -22,6 +24,9 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
|
|
|
22
24
|
isTty = Boolean(process.stdin.isTTY && process.stdout.isTTY),
|
|
23
25
|
running = listRunningSandboxes,
|
|
24
26
|
launch = launchSandbox,
|
|
27
|
+
// The docker spawn used for this session's egress network, seamed like `launch` so the tests never
|
|
28
|
+
// touch a daemon. Not used when PI_EGRESS=0.
|
|
29
|
+
spawnNetwork = spawn,
|
|
25
30
|
now = () => Date.now(),
|
|
26
31
|
} = deps;
|
|
27
32
|
|
|
@@ -89,6 +94,10 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
|
|
|
89
94
|
else err(`warning: could not pin ${jobId}: ${pinned.reason}\n`);
|
|
90
95
|
}
|
|
91
96
|
|
|
97
|
+
// REQ-EGRESS-ALLOWLIST: this session's own network, exactly like a job's, named off its own container
|
|
98
|
+
// so the reaper's `pi-job-` filter never touches it -- a worker restart must not tear the network out
|
|
99
|
+
// from under a shell an operator is sitting in.
|
|
100
|
+
const network = config.egress ? networkNameFor(resolved.name) : null;
|
|
92
101
|
const args = buildSandboxRunArgs({
|
|
93
102
|
image: resolved.manifest.image,
|
|
94
103
|
name: resolved.name,
|
|
@@ -97,15 +106,27 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
|
|
|
97
106
|
publish,
|
|
98
107
|
term: env.TERM,
|
|
99
108
|
idleSeconds: config.sandboxIdleMinutes * 60,
|
|
109
|
+
network,
|
|
110
|
+
egressEnv: egressEnv({ proxy: config.egressProxy, armed: config.egress }),
|
|
100
111
|
});
|
|
101
112
|
|
|
102
113
|
out(`opening ${resolved.name} — image ${resolved.manifest.image}, workspace ${resolved.manifest.workspace}\n`);
|
|
103
114
|
out("no credentials are set in this container. exit the shell to dispose of it.\n");
|
|
104
115
|
if (publish.length > 0) out(`published: ${publish.filter((f) => f !== "-p").join(", ")}\n`);
|
|
105
116
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
117
|
+
// No pre-spend gate here, deliberately: that is a MONEY gate and a sandbox spends nothing. A missing
|
|
118
|
+
// proxy fails at `docker run` with docker's own message, in front of an operator at a terminal, which
|
|
119
|
+
// is the one place a late failure is cheap.
|
|
120
|
+
if (network && !(await createJobNetwork(spawnNetwork, { network, proxy: config.egressProxy }))) {
|
|
121
|
+
return fail(err, `could not create the egress network ${network} -- is the proxy running? \`docker compose -f deploy/docker-compose.yml --profile egress up -d\``);
|
|
122
|
+
}
|
|
123
|
+
try {
|
|
124
|
+
const { code, error } = await launch({ args });
|
|
125
|
+
if (error) return fail(err, `could not start docker: ${error.message}`);
|
|
126
|
+
return code ?? 0;
|
|
127
|
+
} finally {
|
|
128
|
+
if (network) await removeJobNetwork(spawnNetwork, { network, proxy: config.egressProxy });
|
|
129
|
+
}
|
|
109
130
|
}
|
|
110
131
|
|
|
111
132
|
/**
|
package/src/sandbox.mjs
CHANGED
|
@@ -79,8 +79,10 @@ function inPortRange(n) {
|
|
|
79
79
|
* @param publish already-parsed `-p` flags
|
|
80
80
|
* @param term the host's TERM, so the shell renders
|
|
81
81
|
* @param idleSeconds bash's own TMOUT; 0 omits it
|
|
82
|
+
* @param network this session's own egress network (REQ-EGRESS-ALLOWLIST); null = the default bridge
|
|
83
|
+
* @param egressEnv the proxy variables that go with it, or {} when no policy is armed
|
|
82
84
|
*/
|
|
83
|
-
export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish = [], term, idleSeconds = 0 }) {
|
|
85
|
+
export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish = [], term, idleSeconds = 0, network = null, egressEnv: proxyEnv = {} }) {
|
|
84
86
|
return buildDockerRunArgs({
|
|
85
87
|
image,
|
|
86
88
|
name,
|
|
@@ -89,9 +91,20 @@ export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish =
|
|
|
89
91
|
// The ONLY two variables, and neither is a credential. TERM so the shell renders; TMOUT so a
|
|
90
92
|
// forgotten session closes itself. `buildDockerRunArgs` skips undefined, so an unset TERM or a
|
|
91
93
|
// disabled idle timeout emits nothing rather than an empty string.
|
|
94
|
+
// A sandbox joins the SAME kind of network a job did, by the same builder, so the boundary cannot
|
|
95
|
+
// land on job containers and miss this one. Leaving sandboxes on the default bridge was the tempting
|
|
96
|
+
// alternative and it is the wrong one: it reads as a convenience (install a missing dependency while
|
|
97
|
+
// debugging) and it is a WIDER reach than the run the sandbox exists to reproduce. A shell that can
|
|
98
|
+
// go where the run could not is not reproducing the run. Nothing an operator wants is lost, because
|
|
99
|
+
// the forge and the registry are on the allowlist a job needed anyway.
|
|
100
|
+
network,
|
|
92
101
|
env: {
|
|
93
102
|
TERM: term || undefined,
|
|
94
103
|
TMOUT: idleSeconds > 0 ? String(idleSeconds) : undefined,
|
|
104
|
+
// Still NO CREDENTIALS, and that clause is untouched: a proxy URL is not a credential, and
|
|
105
|
+
// buildContainerEnv is still not reused here. The env is two variables about the terminal and,
|
|
106
|
+
// when a policy is armed, three about the network.
|
|
107
|
+
...proxyEnv,
|
|
95
108
|
},
|
|
96
109
|
// Ahead of the env and the mounts, and well ahead of the image, which buildDockerRunArgs keeps as
|
|
97
110
|
// the final positional. `--entrypoint` also clears the image's CMD; this repo's Dockerfile sets
|