@edgehero/pi-dispatch 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/processor.mjs CHANGED
@@ -41,6 +41,9 @@ export async function runJob(job, deps) {
41
41
  // this job names is on this host (image-preflight.mjs). Default admits everything, so a wiring that
42
42
  // omits it behaves exactly as before -- the container's own failure stays the backstop.
43
43
  imagePreflight = async () => ({ ok: true }),
44
+ // REQ-EGRESS-ALLOWLIST. Default admits everything, so a wiring that omits it behaves exactly as a
45
+ // deployment with no egress policy does -- which is also what the real factory returns when unarmed.
46
+ egressPreflight = async () => ({ ok: true }),
44
47
  // (session, { piVersion }) => { promoted, reason, bytes }. Promotes this job's transcript back into
45
48
  // the store, on a COMPLETED exit only. Never throws. The default is a no-op so a wiring that omits
46
49
  // it behaves exactly as before -- no store, no promotion, no session in the record.
@@ -184,6 +187,46 @@ export async function runJob(job, deps) {
184
187
  throw new InfraRetry("docker unavailable, image preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
185
188
  }
186
189
 
190
+ // REQ-EGRESS-ALLOWLIST. The egress policy this deployment claims must be able to serve this job
191
+ // BEFORE the job costs anything. It is one `docker inspect` when the policy is armed and ZERO spawns
192
+ // when it is not, so a deployment without one pays nothing at all.
193
+ //
194
+ // PLACEMENT, and it is the same ladder the image preflight sits at the top of. A missing proxy blocks
195
+ // EVERY job of EVERY kind on this host -- like a missing image -- and unlike a missing image it blocks
196
+ // them EXPENSIVELY: the container starts, the provider is unreachable, the runner exits 1, exit 1 is
197
+ // the retryable class, `attempts: 2`, and `releaseBudget` refunds only `container-never-started` --
198
+ // this container started. So each such job spends two job-count slots and buys nothing with either,
199
+ // and a cron-driven deployment empties its daily cap before anyone reads the first failure. That cost
200
+ // is what makes this a pre-spend gate rather than a doc: measured at three provider attempts,
201
+ // `Request timed out.`, exit 1, ~40 seconds, zero tokens (docs/egress.md).
202
+ //
203
+ // A RETURN, never a throw (CONST-RETRY-INFRA-ONLY): retrying never makes an absent proxy appear.
204
+ const egress = await egressPreflight(job);
205
+ if (egress.proxyMissing || egress.proxyStopped) {
206
+ const proxy = egress.proxyMissing ?? egress.proxyStopped;
207
+ const state = egress.proxyMissing ? "is not on this host" : "is not running";
208
+ await comment(job, `Refused: this deployment runs jobs behind an egress policy and its allowlist proxy "${proxy}" ${state}, so the job could not reach the provider and would burn its budget slot proving it. Start it with \`docker compose -f deploy/docker-compose.yml --profile egress up -d\`, or set PI_EGRESS=0 to run without an egress policy. Not run.`);
209
+ // The proxy's NAME is operator-authored deployment config, never payload -- the same PII class as
210
+ // the image ref on the refusal above.
211
+ log(egress.proxyMissing ? "refused_egress_proxy_missing" : "refused_egress_proxy_stopped", { proxy });
212
+ return {
213
+ outcome: "policy",
214
+ reason: egress.proxyMissing ? "egress-proxy-missing" : "egress-proxy-stopped",
215
+ exitCode: null,
216
+ turns: null,
217
+ tokens: null,
218
+ provider: job.provider ?? null,
219
+ model: job.model ?? null,
220
+ budgetReserved: false, // refused before reserveBudget, so no job-count slot was consumed
221
+ };
222
+ }
223
+ if (egress.unavailable) {
224
+ // The daemon did not answer, so this is indeterminate rather than a refusal -- the same
225
+ // determinate/indeterminate split the image preflight draws one gate up, and thrown for the same
226
+ // reason. Pre-reserve, so the refund below is a no-op and still honest if this gate ever moves.
227
+ throw new InfraRetry("docker unavailable, egress preflight could not run", { reason: "container-never-started", provider: job.provider ?? null, model: job.model ?? null });
228
+ }
229
+
187
230
  // REQ-RESUMABLE-SESSION's one fail-CLOSED case. Everything else in that feature fails OPEN and
188
231
  // NAMES itself -- absent, expired, too-large, unparseable, locked, promote-failed -- because a cold
189
232
  // start is a correct run. This one cannot be: with no `sessionsDir`, resolveSession returns null
@@ -1,5 +1,6 @@
1
1
  import { spawn } from "node:child_process";
2
2
  import { buildDockerRunArgs, CONTAINER_SESSION_FILE } from "./docker-run.mjs";
3
+ import { createJobNetwork, networkNameFor, removeJobNetwork } from "./egress.mjs";
3
4
  import { buildContainerEnv } from "./env-allowlist.mjs";
4
5
  import { resolveJobImage } from "./image-preflight.mjs";
5
6
  import { InfraRetry } from "./processor.mjs";
@@ -40,6 +41,8 @@ export function makeRunContainer({
40
41
  forwardEnv = [],
41
42
  authFromPi = false, // fall back to ~/.pi/agent/auth.json for the provider key when the env has none
42
43
  forgeHosts = {}, // per-forge self-hosted instance URLs, so a forge CLI in the container talks to the right one
44
+ egress = false, // REQ-EGRESS-ALLOWLIST: put this job on its own --internal network behind the allowlist proxy
45
+ egressProxy, // the proxy component attached to that network; undefined = egress.mjs's default name
43
46
  }) {
44
47
  // async so a synchronous throw (e.g. buildContainerEnv on an unconfigured provider) surfaces as
45
48
  // a rejection, uniformly awaitable by the processor and by tests.
@@ -59,6 +62,8 @@ export function makeRunContainer({
59
62
  forgeKind: job?.kind,
60
63
  forgeHosts,
61
64
  hostEnv,
65
+ egress, // REQ-EGRESS-ALLOWLIST: emits HTTPS_PROXY/HTTP_PROXY/NO_PROXY/NODE_USE_ENV_PROXY, or nothing
66
+ egressProxy,
62
67
  allowGlobalExtensions, // REQ-GLOBAL-PI-OVERLAY: false emits the explicit PI_GLOBAL_ALLOW_EXTENSIONS=0 opt-out
63
68
  // REQ-GLOBAL-PI-OVERLAY: the per-job value comes off `job` (like maxTurns), the staged set off
64
69
  // the closure (like allowGlobalExtensions) -- so a trigger can withhold what the operator staged.
@@ -84,6 +89,10 @@ export function makeRunContainer({
84
89
  authFromPi, // source the provider key from pi's auth.json when the env has none
85
90
  });
86
91
 
92
+ // `-net` on this container's own name (egress.mjs). null when no policy is armed, and docker-run's
93
+ // guard then omits the flag entirely, so the argv is byte-identical to one built before this feature.
94
+ const network = egress ? networkNameFor(name) : null;
95
+
87
96
  const args = buildDockerRunArgs({
88
97
  // Same split as packagePaths above: the per-job value off `job`, the deployment value off the closure,
89
98
  // so a trigger can name its own toolchain (INT-TRIGGERS-FILE-CONTRACT). Resolved through the SAME
@@ -99,13 +108,26 @@ export function makeRunContainer({
99
108
  sessionDir: prepared.session?.hostDir,
100
109
  globalPiDir, // undefined/null -> docker-run's guard skips the /opt/pi-global mount
101
110
  name,
111
+ network, // REQ-EGRESS-ALLOWLIST: null when no policy is armed, and the flag is then absent
102
112
  });
103
113
 
114
+ // REQ-EGRESS-ALLOWLIST. This job's own --internal network, created here rather than at boot because
115
+ // it holds exactly two endpoints -- this container and the proxy -- and that is what makes job-to-job
116
+ // traffic structurally impossible rather than merely discouraged. A shared network could not do it:
117
+ // `enable_icc=false` would block job-to-job AND job-to-proxy, since ICC governs every container pair
118
+ // on the bridge and the proxy is a container.
119
+ //
120
+ // A failure to build it is INFRA, not policy: nothing has been spent, a retry may well succeed, and
121
+ // `container-never-started` is literally true, so the reservation is given back (processor.mjs).
122
+ if (network && !(await createJobNetwork(spawnFn, { network, proxy: egressProxy }))) {
123
+ throw new InfraRetry("container-never-started", { reason: "container-never-started" });
124
+ }
125
+
104
126
  // Host-side per-job log sink, teed off `onOutput`. `name` is `pi-job-<jobId>`; the sink
105
127
  // sanitizes internally. No container mount, no env var -- the sink lives on this side only.
106
128
  const sink = openJobLog(name);
107
129
 
108
- return await new Promise((resolve, reject) => {
130
+ const run = new Promise((resolve, reject) => {
109
131
  const child = spawnFn("docker", args, { stdio: ["ignore", "pipe", "pipe"] });
110
132
  // A throwing sink.write is swallowed so a misbehaving sink cannot break the tee or hang the run.
111
133
  const tee = (chunk) => {
@@ -141,5 +163,14 @@ export function makeRunContainer({
141
163
  resolve(aborted ? { code: code ?? 137, aborted: true, turns, tokens, session, usage } : { code: code ?? 1, aborted: false, turns, tokens, session, usage });
142
164
  });
143
165
  });
166
+
167
+ // The network outlives the container by exactly this `finally`. Best-effort and never throwing: the
168
+ // container has already exited, its code is the job's answer, and a teardown fault must not rewrite
169
+ // that answer. What a failure leaves behind is a memberless network, which the boot reaper sweeps.
170
+ try {
171
+ return await run;
172
+ } finally {
173
+ if (network) await removeJobNetwork(spawnFn, { network, proxy: egressProxy });
174
+ }
144
175
  };
145
176
  }
@@ -246,7 +246,7 @@ function rebuildUsage(u) {
246
246
  * path embeds the operator's OS account name.
247
247
  *
248
248
  * `reason` is a fixed enum passthrough (worker-abort | over-budget | unprotected-branch |
249
- * runner-policy | job-image-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
249
+ * runner-policy | job-image-missing | egress-proxy-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
250
250
  * default to `null` when the outcome does not carry them, so the record shape is stable whether or not
251
251
  * the source reports those fields.
252
252
  */
@@ -1,6 +1,8 @@
1
+ import { spawn } from "node:child_process";
1
2
  import { parseArgs } from "node:util";
2
3
  import { loadConfig } from "./config.mjs";
3
4
  import { sanitizeJobId } from "./run-history.mjs";
5
+ import { createJobNetwork, egressEnv, networkNameFor, removeJobNetwork } from "./egress.mjs";
4
6
  import { buildSandboxRunArgs, launchSandbox, listRunningSandboxes, parsePublish, resolveSandbox, sandboxContainerName } from "./sandbox.mjs";
5
7
  import { listSandboxes, pinSandbox } from "./sandbox-store.mjs";
6
8
 
@@ -22,6 +24,9 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
22
24
  isTty = Boolean(process.stdin.isTTY && process.stdout.isTTY),
23
25
  running = listRunningSandboxes,
24
26
  launch = launchSandbox,
27
+ // The docker spawn used for this session's egress network, seamed like `launch` so the tests never
28
+ // touch a daemon. Not used when PI_EGRESS=0.
29
+ spawnNetwork = spawn,
25
30
  now = () => Date.now(),
26
31
  } = deps;
27
32
 
@@ -89,6 +94,10 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
89
94
  else err(`warning: could not pin ${jobId}: ${pinned.reason}\n`);
90
95
  }
91
96
 
97
+ // REQ-EGRESS-ALLOWLIST: this session's own network, exactly like a job's, named off its own container
98
+ // so the reaper's `pi-job-` filter never touches it -- a worker restart must not tear the network out
99
+ // from under a shell an operator is sitting in.
100
+ const network = config.egress ? networkNameFor(resolved.name) : null;
92
101
  const args = buildSandboxRunArgs({
93
102
  image: resolved.manifest.image,
94
103
  name: resolved.name,
@@ -97,15 +106,27 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
97
106
  publish,
98
107
  term: env.TERM,
99
108
  idleSeconds: config.sandboxIdleMinutes * 60,
109
+ network,
110
+ egressEnv: egressEnv({ proxy: config.egressProxy, armed: config.egress }),
100
111
  });
101
112
 
102
113
  out(`opening ${resolved.name} — image ${resolved.manifest.image}, workspace ${resolved.manifest.workspace}\n`);
103
114
  out("no credentials are set in this container. exit the shell to dispose of it.\n");
104
115
  if (publish.length > 0) out(`published: ${publish.filter((f) => f !== "-p").join(", ")}\n`);
105
116
 
106
- const { code, error } = await launch({ args });
107
- if (error) return fail(err, `could not start docker: ${error.message}`);
108
- return code ?? 0;
117
+ // No pre-spend gate here, deliberately: that is a MONEY gate and a sandbox spends nothing. A missing
118
+ // proxy fails at `docker run` with docker's own message, in front of an operator at a terminal, which
119
+ // is the one place a late failure is cheap.
120
+ if (network && !(await createJobNetwork(spawnNetwork, { network, proxy: config.egressProxy }))) {
121
+ return fail(err, `could not create the egress network ${network} -- is the proxy running? \`docker compose -f deploy/docker-compose.yml --profile egress up -d\``);
122
+ }
123
+ try {
124
+ const { code, error } = await launch({ args });
125
+ if (error) return fail(err, `could not start docker: ${error.message}`);
126
+ return code ?? 0;
127
+ } finally {
128
+ if (network) await removeJobNetwork(spawnNetwork, { network, proxy: config.egressProxy });
129
+ }
109
130
  }
110
131
 
111
132
  /**
package/src/sandbox.mjs CHANGED
@@ -79,8 +79,10 @@ function inPortRange(n) {
79
79
  * @param publish already-parsed `-p` flags
80
80
  * @param term the host's TERM, so the shell renders
81
81
  * @param idleSeconds bash's own TMOUT; 0 omits it
82
+ * @param network this session's own egress network (REQ-EGRESS-ALLOWLIST); null = the default bridge
83
+ * @param egressEnv the proxy variables that go with it, or {} when no policy is armed
82
84
  */
83
- export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish = [], term, idleSeconds = 0 }) {
85
+ export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish = [], term, idleSeconds = 0, network = null, egressEnv: proxyEnv = {} }) {
84
86
  return buildDockerRunArgs({
85
87
  image,
86
88
  name,
@@ -89,9 +91,20 @@ export function buildSandboxRunArgs({ image, name, workspace, jobDir, publish =
89
91
  // The ONLY two variables, and neither is a credential. TERM so the shell renders; TMOUT so a
90
92
  // forgotten session closes itself. `buildDockerRunArgs` skips undefined, so an unset TERM or a
91
93
  // disabled idle timeout emits nothing rather than an empty string.
94
+ // A sandbox joins the SAME kind of network a job did, by the same builder, so the boundary cannot
95
+ // land on job containers and miss this one. Leaving sandboxes on the default bridge was the tempting
96
+ // alternative and it is the wrong one: it reads as a convenience (install a missing dependency while
97
+ // debugging) and it is a WIDER reach than the run the sandbox exists to reproduce. A shell that can
98
+ // go where the run could not is not reproducing the run. Nothing an operator wants is lost, because
99
+ // the forge and the registry are on the allowlist a job needed anyway.
100
+ network,
92
101
  env: {
93
102
  TERM: term || undefined,
94
103
  TMOUT: idleSeconds > 0 ? String(idleSeconds) : undefined,
104
+ // Still NO CREDENTIALS, and that clause is untouched: a proxy URL is not a credential, and
105
+ // buildContainerEnv is still not reused here. The env is two variables about the terminal and,
106
+ // when a policy is armed, three about the network.
107
+ ...proxyEnv,
95
108
  },
96
109
  // Ahead of the env and the mounts, and well ahead of the image, which buildDockerRunArgs keeps as
97
110
  // the final positional. `--entrypoint` also clears the image's CMD; this repo's Dockerfile sets