@edgehero/pi-dispatch 1.9.0 → 1.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/doctor.mjs CHANGED
@@ -58,6 +58,7 @@ import { agentDirFrom, readHostPi } from "./host-pi.mjs";
58
58
  import { PACKAGES_SUBDIR, readStagedSkills, readStageManifest } from "./packages.mjs";
59
59
  import { copySkillTree } from "./copy-tree.mjs";
60
60
  import { SKILL_NAME_RE } from "./flow-gate.mjs";
61
+ import { ABSENT, ASSERTED, PROPERTY_NAMES, declarationOf, floorShortfall, parseBackendFloor, parseBackendList, unarmedFloor } from "./backends.mjs";
61
62
  import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
62
63
  import { installedUnitPaths, readUnitSeam } from "./service.mjs";
63
64
  import { parseSecretProfiles } from "./secret-profiles.mjs";
@@ -469,6 +470,7 @@ export async function collectChecks(env, seams) {
469
470
  // byte-identical output. Gated on docker and the image, because two of these checks run a container and
470
471
  // the rest are noise on top of a down daemon.
471
472
  checks.push(...(await egressChecks(env, seams, { dockerCode, imageCode, jobImage })));
473
+ checks.push(...backendChecks(env));
472
474
 
473
475
  // The receiver itself, when the triggers file names ANY forge (issue #80). Only forge deliveries need
474
476
  // the receiver at all, so a cron/local-only deployment gets no receiver noise here. WARNS rather than
@@ -2337,3 +2339,119 @@ async function defaultProbeValkey(url) {
2337
2339
  client.disconnect();
2338
2340
  }
2339
2341
  }
2342
+
2343
+ /**
2344
+ * WHERE this deployment's jobs run, and what that place actually guarantees (issue #227).
2345
+ *
2346
+ * THIS IS THE CHECK THAT MAKES THE DECLARATION ADMISSIBLE AT ALL. `CONST-EGRESS-POLICY-IN-THE-ARGV` says a
2347
+ * control an operator BELIEVES in is worse than one they know is missing, because the belief displaces the
2348
+ * credential bound that is really holding. A table of guarantees nothing ever prints is exactly such a
2349
+ * belief. So the three words must stay TOLD APART on the way out, and told apart ON THE SCREEN rather than
2350
+ * in a field nobody renders:
2351
+ *
2352
+ * enforced -- ours, in this worker's own code, readable back from what it produced. Quiet.
2353
+ * asserted -- someone else's. Rendered as a WARNING, and it NAMES who is asserting it, because "not us"
2354
+ * without "them" leaves an operator nothing to go and check.
2355
+ * absent -- not provided at all. A failure, since a deployment reaching it must know before a job does.
2356
+ *
2357
+ * `ok: false, warn: true` IS THE WARNING SHAPE, and it is the one thing to get right when editing here.
2358
+ * `render` reads `c.ok` FIRST, so `ok: true, warn: true` renders as a plain pass and drops the `fix` line
2359
+ * with it. An earlier draft used that shape and every asserted property printed as a green tick, which made
2360
+ * this section say the opposite of what it exists to say. `warn` keeps the RUN green -- `render` only fails
2361
+ * on `!ok && !warn` -- so an operator's CI is unaffected while the operator is actually told.
2362
+ *
2363
+ * A property a deployment switch gates is printed with the switch AND its position, never the bare
2364
+ * capability word: `local` can enforce egress, and a `PI_EGRESS=0` deployment is not getting it. Those are
2365
+ * two different sentences. `absent` OUTRANKS the gate, because a control that does not exist is a different
2366
+ * fact from one that is merely unarmed, and "CAN be absent but the switch is off" would be both meaningless
2367
+ * and green.
2368
+ *
2369
+ * Reads the environment directly, like every other check here, and parses through `backends.mjs` so doctor
2370
+ * and the worker cannot disagree about what a floor says.
2371
+ */
2372
+ export function backendChecks(env) {
2373
+ const checks = [];
2374
+ let backends;
2375
+ let floor;
2376
+ try {
2377
+ backends = parseBackendList(env.PI_BACKENDS);
2378
+ floor = parseBackendFloor(env.PI_BACKEND_FLOOR);
2379
+ } catch (error) {
2380
+ // The worker refuses to boot on this, so doctor must not soften it to a warning.
2381
+ return [{ ok: false, label: `backend configuration does not parse: ${error.message}`, fix: "fix PI_BACKENDS / PI_BACKEND_FLOOR, then re-run doctor" }];
2382
+ }
2383
+
2384
+ // The switch positions every `armedBy` in the table can name. A MAP rather than one boolean, because
2385
+ // `armedBy` is a general field: hardcoding one variable name here would silently hide a second switch's
2386
+ // off-position the day one is added, which is the defect `armedBy` exists to prevent.
2387
+ const switches = {};
2388
+ try {
2389
+ switches.PI_EGRESS = egressArmed(env);
2390
+ } catch (error) {
2391
+ // NOT an abstention that falls through to the good case. A value doctor cannot parse is a value the
2392
+ // worker refuses to boot on, and an earlier draft claimed in a comment that "its own check reports
2393
+ // that" -- nothing did, so doctor printed every gated property as quietly enforced on a deployment
2394
+ // that could not start.
2395
+ checks.push({ ok: false, label: `PI_EGRESS does not parse, so what this deployment actually gets cannot be determined: ${error.message}`, fix: 'set PI_EGRESS to exactly "0" (off) or "1"/unset (on)' });
2396
+ }
2397
+
2398
+ // The dispatch landed in slice 4, so the "nothing selects yet" qualifier came off -- and it came off HERE
2399
+ // as well as in the code, because a stale caveat on the one surface that makes the table admissible is
2400
+ // its own kind of false statement.
2401
+ checks.push({ ok: true, label: `Jobs run on: ${backends.join(", ")}${backends.length > 1 ? ` (a trigger that names none runs on ${backends[0]}; run.backend selects)` : ""}` });
2402
+
2403
+ for (const name of backends) {
2404
+ for (const property of PROPERTY_NAMES) {
2405
+ const d = declarationOf(name, property);
2406
+ if (!d) continue;
2407
+ // FIRST, ahead of the gate: a control that does not exist is not a control that is unarmed.
2408
+ if (d.word === ABSENT) {
2409
+ checks.push({ ok: false, label: `${name}: ${property} is ABSENT -- ${d.question}`, fix: `this backend does not provide ${property}; a deployment that needs it must not run jobs on ${name}` });
2410
+ continue;
2411
+ }
2412
+ if (d.armedBy && switches[d.armedBy] === undefined) {
2413
+ // The switch did not parse. Say so rather than pick a side; the failure is already reported.
2414
+ checks.push({ ok: false, warn: true, label: `${name}: ${property} depends on ${d.armedBy}, which does not parse -- cannot say whether this deployment gets it`, fix: `fix ${d.armedBy}, then re-run doctor` });
2415
+ continue;
2416
+ }
2417
+ if (d.armedBy && switches[d.armedBy] === false) {
2418
+ checks.push({ ok: false, warn: true, label: `${name}: ${property} CAN be ${d.word} here, but ${d.armedBy} is off, so this deployment is not getting it`, fix: `arm ${d.armedBy} to get it (${d.question})` });
2419
+ continue;
2420
+ }
2421
+ if (d.word === ASSERTED) {
2422
+ checks.push({ ok: false, warn: true, label: `${name}: ${property} is ASSERTED by ${d.assertedBy ?? "something outside this worker"}, not enforced by it`, fix: `not verifiable from here, so treat it as a claim rather than a control: ${d.question}` });
2423
+ continue;
2424
+ }
2425
+ // enforced, and armed if it is gated at all. The good case, and it stays quiet.
2426
+ }
2427
+ }
2428
+
2429
+ // ALWAYS a line, including when no floor is set. `PI_BACKENDS_FLOOR` is a plausible one-character-off
2430
+ // spelling of the real name, and nothing in this project warns on an unknown PI_* variable, so silence
2431
+ // here would make a typo'd VARIABLE NAME look exactly like a floor that holds -- the same belief the
2432
+ // strict parsing inside the string exists to prevent, arriving from outside the string.
2433
+ const floorNames = Object.keys(floor);
2434
+ if (floorNames.length === 0) {
2435
+ checks.push({ ok: true, label: "PI_BACKEND_FLOOR is not set, so no minimum is required of any backend" });
2436
+ return checks;
2437
+ }
2438
+
2439
+ const misses = floorShortfall(backends, floor);
2440
+ const unarmed = unarmedFloor(floor, switches);
2441
+ // A floor whose every entry is `absent` parses, reads, and bounds NOTHING: `meets(have, absent)` is true
2442
+ // for every value. It is the one READABLE word that reproduces the outcome `isDeclaration` refuses a
2443
+ // typo for, so it is named rather than affirmed.
2444
+ const bounding = floorNames.filter((p) => floor[p] !== ABSENT);
2445
+ const spelled = floorNames.map((p) => `${p}=${floor[p]}`).join(", ");
2446
+ if (misses.length > 0) {
2447
+ checks.push({ ok: false, label: `PI_BACKEND_FLOOR is not met: ${misses.map((m) => `${m.backend}.${m.property} is ${m.have}`).join(", ")}`, fix: "raise the backend, lower PI_BACKEND_FLOOR, or drop the backend from PI_BACKENDS" });
2448
+ } else if (unarmed.length > 0) {
2449
+ checks.push({ ok: false, label: `PI_BACKEND_FLOOR asks for ${unarmed.map((u) => `${u.property}=${u.want}`).join(", ")}, which ${[...new Set(unarmed.map((u) => u.armedBy))].join(", ")} has switched off`, fix: "arm the switch, or lower that entry to `absent` if you did not mean to require it" });
2450
+ } else if (bounding.length === 0) {
2451
+ checks.push({ ok: false, warn: true, label: `PI_BACKEND_FLOOR (${spelled}) requires nothing: every entry asks for "absent", which every backend meets`, fix: "raise an entry to `asserted` or `enforced` for it to bound anything" });
2452
+ } else {
2453
+ checks.push({ ok: true, label: `PI_BACKEND_FLOOR holds (${spelled})` });
2454
+ }
2455
+
2456
+ return checks;
2457
+ }
package/src/index.mjs CHANGED
@@ -1,13 +1,11 @@
1
- import { execFile } from "node:child_process";
2
- import { promisify } from "node:util";
3
1
  import { DelayedError, UnrecoverableError, Worker } from "bullmq";
2
+ import { jobContainerName } from "./backend-local.mjs";
4
3
  import { InfraRetry, runJob } from "./processor.mjs";
5
4
  import { targetFor } from "./run-history.mjs";
6
5
  import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
7
6
  import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
8
7
  import { makeWaitState } from "./wait-state.mjs";
9
8
 
10
- const exec = promisify(execFile);
11
9
 
12
10
  export const QUEUE = "pi-jobs";
13
11
 
@@ -44,6 +42,50 @@ const THROTTLE_ALARM = 5;
44
42
  // the only evidence there is. A test pins the two apart.
45
43
  const THROTTLE_FLOOR_MS = 11_000;
46
44
 
45
+ /**
46
+ * How long a container gets to actually die after the abort's `docker stop` before the worker stops waiting.
47
+ *
48
+ * `docker stop -t 5` is SIGTERM then an unignorable SIGKILL five seconds later, so a reachable daemon ends
49
+ * the container well inside this. The margin is for the daemon being slow, not for the container being
50
+ * stubborn -- a container cannot outlive SIGKILL.
51
+ */
52
+ const ABORT_GRACE_MS = 30_000;
53
+
54
+ /**
55
+ * Resolve `run` normally, but stop waiting once the abort has fired and the grace has passed.
56
+ *
57
+ * See the call site for why this exists. Returns the same `{ code: 137, aborted: true }` shape a killed
58
+ * container produces, so nothing downstream needs to know the difference -- the processor's abort
59
+ * classification, the run record and the refund all behave exactly as they do for a stop that worked.
60
+ */
61
+ function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
62
+ if (!signal) return run;
63
+ return new Promise((resolve, reject) => {
64
+ let timer = null;
65
+ let settled = false;
66
+ const done = (fn) => (v) => {
67
+ if (settled) return;
68
+ settled = true;
69
+ clearTimeout(timer);
70
+ fn(v);
71
+ };
72
+ const onAbort = () => {
73
+ timer = setTimeout(() => {
74
+ if (settled) return;
75
+ settled = true;
76
+ log("stop_did_not_take", { job: job.id, graceMs });
77
+ resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null });
78
+ }, graceMs);
79
+ // A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
80
+ // has finished, and this timer only matters while a job is still in flight.
81
+ timer.unref?.();
82
+ };
83
+ if (signal.aborted) onAbort();
84
+ else signal.addEventListener("abort", onAbort, { once: true });
85
+ run.then(done(resolve), done(reject));
86
+ });
87
+ }
88
+
47
89
  /**
48
90
  * Build the BullMQ processor.
49
91
  *
@@ -53,7 +95,8 @@ const THROTTLE_FLOOR_MS = 11_000;
53
95
  * abort -- with no error. A test asserts the arity precisely because the failure is silent.
54
96
  *
55
97
  * Dependencies are injected so this is testable without a live queue: `cancelJob` (fired by the
56
- * timeout), `stopContainer` (fired by the abort), and the orchestration deps.
98
+ * timeout), `stopContainer` (fired by the abort, and INJECTED so the venue that built the container is
99
+ * the one that stops it), and the orchestration deps.
57
100
  *
58
101
  * Once per job, before runJob, it resolves the runtime-settings overlay via `getSettings`
59
102
  * (INT-CONFIG-OVERLAY-CONTRACT). A present-but-invalid overlay resolves to a POLICY refusal RETURNED
@@ -63,7 +106,7 @@ const THROTTLE_FLOOR_MS = 11_000;
63
106
  * The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
64
107
  * runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
65
108
  */
66
- export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
109
+ export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
67
110
  return async function processor(job, token, signal) {
68
111
  // Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
69
112
  // window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
@@ -491,6 +534,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
491
534
  let name;
492
535
  let timer;
493
536
  let onAbort;
537
+ let venue;
494
538
  try {
495
539
  // Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
496
540
  // finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
@@ -498,7 +542,23 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
498
542
  // addEventListener on the bullmq-allocated controller are total at processor arity 3); the
499
543
  // guard is structural, not observational.
500
544
  startedAt = new Date().toISOString();
501
- name = `pi-job-${job.id}`;
545
+ // The producer of the name both boot reapers sweep by substring. Built from the shared prefix
546
+ // rather than typed here, so a rename cannot land in the producer and not in the sweeps (#227).
547
+ // From the VENUE that will build the container, not from the local adapter reached for directly:
548
+ // the abort stops this name, so the name and the stop have to come from the same backend. The
549
+ // default keeps every wiring that predates the seam building it exactly as before.
550
+ //
551
+ // `job.data`, NOT `job`. This function's `job` is the BullMQ WRAPPER -- its own keys are `id` and
552
+ // `data` -- so `job.backend` is always undefined and the registry's resolution would fall to
553
+ // `?? defaultName` for every job, silently, on the one path where the fail-closed throw can never
554
+ // fire because "names nothing" is exactly the case it permits. `runJob` is handed `effectiveJob`,
555
+ // a spread of `job.data`, which is why `runContainer` and the two preflights dispatch correctly
556
+ // while these two did not.
557
+ // One object carrying BOTH halves: the id is the BullMQ wrapper's (it always was) and the venue is
558
+ // the trigger's, which lives in `data`. Built once so the name and the stop cannot resolve
559
+ // different backends -- the whole reason the name moved onto the registry in the first place.
560
+ venue = { ...job.data, id: job.id };
561
+ name = containerName(venue);
502
562
  timer = setTimeout(() => {
503
563
  // BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
504
564
  Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
@@ -507,7 +567,23 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
507
567
  // Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
508
568
  // after the grace period; the runner exits and runContainer returns/throws.
509
569
  onAbort = () => {
510
- Promise.resolve(stopContainer(name)).catch(() => {});
570
+ // The JOB'S DATA goes with the name -- `job.data`, not `job`. A container name alone cannot say
571
+ // which runtime holds it once there is more than one venue, and this call is the only thing
572
+ // standing between a runaway job and the 30-minute bound (REQ-JOB-TIMEOUT-30M). Passing the
573
+ // BullMQ wrapper sent every abort to the DEFAULT venue: `docker stop` on a host that never
574
+ // held the container, rejecting into a log line while the real one kept running and kept
575
+ // spending, with the local reaper unable to see it either.
576
+ // The whole call is inside the try, not just its promise. `Promise.resolve(f())` evaluates `f()`
577
+ // FIRST, so a missing or throwing `stopContainer` raises synchronously, inside an
578
+ // AbortSignal listener, where it surfaces as an uncaughtException and takes the worker
579
+ // process down 30 minutes into a runaway job -- killing every other in-flight job on the
580
+ // host. Losing the kill for one job is bad; losing the process is worse.
581
+ const note = deps.log ?? (() => {});
582
+ try {
583
+ Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: err?.message }));
584
+ } catch (err) {
585
+ note("stop_container_failed", { job: job.id, reason: err?.message });
586
+ }
511
587
  };
512
588
  signal.addEventListener("abort", onAbort, { once: true });
513
589
  } catch (error) {
@@ -574,7 +650,26 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
574
650
  // life. Null when no row carries a money window for this scope.
575
651
  scopedCaps: budgetCapsFor(job.data, limits),
576
652
  ...deps,
577
- runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
653
+ // #227. BOUNDED AFTER THE ABORT, and this is what makes `abortable` an honest declaration.
654
+ //
655
+ // `makeRunContainer`'s promise settles ONLY on the docker child's `close` or `error`. Nothing
656
+ // else ends that await -- the signal is read at entry and captured at close, never passed to
657
+ // the spawn. So `stopContainer` is the sole kill channel, and if it does not take (an
658
+ // unreachable daemon, a wiring whose stop is a no-op) the container keeps running, `docker
659
+ // run` never exits, and the processor awaits FOREVER: an active job renewing its lock and
660
+ // holding its in-flight slot, its host slot, its scope lease and its budget reservation, with
661
+ // nothing in the log and nothing in the record. REQ-JOB-TIMEOUT-30M's Acceptance says "the
662
+ // slot is freed", and it was not.
663
+ //
664
+ // So once the abort has fired, the wait is bounded. On expiry this resolves the SAME shape a
665
+ // killed container returns -- `{ code: 137, aborted: true }`, which `run-container.mjs`
666
+ // already uses for "aborted before it could start" -- so the processor classifies it as
667
+ // POLICY and does not retry. That is deliberate: the container may still be running, and a
668
+ // retry would pay for a second one alongside it. What is leaked is the container, which the
669
+ // next boot reaper sweeps; what is NOT leaked is the slot, the lease and the reservation.
670
+ // The `stop_did_not_take` line is the loud half, because a host whose daemon ignores a stop
671
+ // is a fact an operator has to learn from somewhere.
672
+ runContainer: (ctx) => boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})),
578
673
  // REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
579
674
  // to be abortable for the same reason runContainer does: a resolver blocking on an unreachable
580
675
  // vault would otherwise hold its slot until its own timeout, and an abort landing mid-resolution
@@ -598,12 +693,23 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
598
693
  // job.data with no `.id` -- and the real wrapper is only in scope here, so inject it as
599
694
  // `queueJobId`, mirroring the collectChain injection above. Omitted when unwired so a bare
600
695
  // processor keeps runJob's plain (job, token) call.
601
- ...(deps.prepareWorkspace ? { prepareWorkspace: (j, t) => deps.prepareWorkspace(j, t, { queueJobId: job.id }) } : {}),
696
+ // The third argument is EXTENDED, never replaced. `runJob` calls this as
697
+ // `prepareWorkspace(job, token, { piVersion })`, so a wrapper passing only `{ queueJobId }`
698
+ // dropped it -- and `piVersion` defaults to `null`, which `readCanonical` treats as
699
+ // "never resume": `if (piVersion === null) return COLD("pi-version-changed")`. So EVERY
700
+ // `run.resume` cold-started, on every wired worker, while reporting success. The stamp
701
+ // `promoteSession` writes was correct the whole time; the comparison simply never happened.
702
+ // The processor tests inject `prepareWorkspace` directly and never see this wrapper, which is
703
+ // why nothing caught it (REQ-RESUMABLE-SESSION).
704
+ ...(deps.prepareWorkspace ? { prepareWorkspace: (j, t, opts) => deps.prepareWorkspace(j, t, { ...opts, queueJobId: job.id }) } : {}),
602
705
  // The one-shot pre-spend check (issue #231) needs the REAL BullMQ job's `.id` to excuse this
603
706
  // delivery's own earlier attempt -- runJob's effectiveJob has no `.id`, prepareWorkspace's
604
707
  // own injection above states why, and this one mirrors it. Omitted when unwired so a bare
605
708
  // processor keeps runJob's admit-everything default.
606
- ...(deps.checkOnceSpent ? { checkOnceSpent: (j) => deps.checkOnceSpent(j, { queueJobId: job.id }) } : {}),
709
+ // Same extend-don't-replace shape as `prepareWorkspace` above, for its reason: `runJob` passes
710
+ // this one no options today, so there is nothing to lose yet -- and the day it does, a
711
+ // replacing wrapper would lose it in the same silence.
712
+ ...(deps.checkOnceSpent ? { checkOnceSpent: (j, opts) => deps.checkOnceSpent(j, { ...opts, queueJobId: job.id }) } : {}),
607
713
  });
608
714
  recordRun({ job, result, startedAt, endedAt: new Date().toISOString() });
609
715
  return result;
@@ -630,12 +736,19 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
630
736
  };
631
737
  }
632
738
 
633
- export function createWorker({ connection, name, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
739
+ export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
634
740
  // One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
635
741
  // check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
636
742
  // because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
637
743
  // the delayed set is gated on each worker's own `Date.now()`, so the fastest clock wins every hop and a
638
744
  // job that had to reach another host might never get there.
745
+ // #227. REFUSED AT CONSTRUCTION, because the alternative is discovering it 30 minutes into a runaway
746
+ // job. `stopContainer` is the only thing that enforces REQ-JOB-TIMEOUT-30M, and it is what the backend
747
+ // table declares as `abortable: enforced` -- a wiring that omits it would make that declaration false
748
+ // while every test that never aborts stayed green.
749
+ if (typeof stopContainer !== "function") {
750
+ throw new Error("createWorker: stopContainer is required -- it is the only thing that enforces the 30-minute job timeout");
751
+ }
639
752
  const names = hostQueue ? [QUEUE, hostQueue] : [QUEUE];
640
753
  const workers = [];
641
754
 
@@ -655,7 +768,12 @@ export function createWorker({ connection, name, hostQueue = null, checkLease =
655
768
  // Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
656
769
  // and the shared handle could not reach it.
657
770
  cancelJob: (id, reason) => worker.cancelJob(id, reason),
658
- stopContainer: (name) => exec("docker", ["stop", "-t", "5", name]),
771
+ // #227. INJECTED, not built here. This was a one-line `docker stop` literal, which meant the abort
772
+ // path -- the only thing that can end a runaway job -- was the one backend function unreachable
773
+ // from `startWorker`. The wiring now passes the registry's per-job stop, so the container is
774
+ // stopped by whatever venue built it.
775
+ stopContainer,
776
+ containerName,
659
777
  redis,
660
778
  getSettings,
661
779
  // Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
@@ -733,3 +851,9 @@ export function createWorker({ connection, name, hostQueue = null, checkLease =
733
851
 
734
852
  return primary;
735
853
  }
854
+
855
+ /**
856
+ * `boundAfterAbort` with a zero grace, for tests only. The real bound is 30 seconds, which no test can wait
857
+ * for, and the behaviour under test is what happens WHEN the grace expires -- not how long it is.
858
+ */
859
+ export const __boundAfterAbortForTests = (run, signal, job, log) => boundAfterAbort(run, signal, job, log, 0);
package/src/outbox.mjs CHANGED
@@ -171,6 +171,14 @@ export function makeCollectChain({ queue, enqueue = enqueueLocalJob, readFlowGat
171
171
  // (INT-OUTBOX-CONTRACT's explicit-property-reads rule). Undefined stays undefined, so a parent with
172
172
  // no image chains a child whose data is byte-identical to today's.
173
173
  image: job.data?.image,
174
+ // #227, and INHERITED for the reason `image` directly above is: a chained child continues its
175
+ // parent's work, so it belongs in the venue the parent's trigger chose, not silently back on
176
+ // the deployment default. Written down rather than left to inference because this file's
177
+ // convention is that every inherit/don't-inherit decision here carries its reason -- and the
178
+ // distinction it turns on is the same one `secrets` is excluded by: a venue is toolchain, a
179
+ // resolved credential is a capability. The agent cannot choose it any more than it can choose
180
+ // its child's image: the value comes off `job.data`, never off the request file.
181
+ backend: job.data?.backend,
174
182
  // The parent's injected skills follow the child, off validated JOB DATA and never off the
175
183
  // request file (REQ-PER-TRIGGER-SKILLS). Same reason `image` does: a chained child runs the
176
184
  // same operator's flows and, without them, would look up a skill that is not there, write a
package/src/packages.mjs CHANGED
@@ -34,10 +34,13 @@ import { configError } from "./config.mjs";
34
34
  // and the admin block cannot drift between the stager and this validator (doctor.mjs sets the precedent
35
35
  // of importing from import-pi.mjs).
36
36
  import { ADMIN_RE, ENTRY_NAME_RE } from "./import-pi.mjs";
37
- // The container-side mount point is docker-run.mjs's fact -- IMPORTED, never re-typed, so the mount and the
38
- // packages root below cannot drift apart while both test suites stay green. docker-run.mjs is dependency-free
39
- // (it builds an argv array and nothing else), so this costs no cycle and no weight in the admin's bundle.
40
- import { CONTAINER_GLOBAL_PI_DIR } from "./docker-run.mjs";
37
+ // The container-side mount point is container-spec.mjs's fact -- IMPORTED, never re-typed, so the mount and
38
+ // the packages root below cannot drift apart while both test suites stay green. That module is a LEAF and
39
+ // imports nothing, so this costs no cycle and no weight in the admin's bundle. It used to be imported from
40
+ // docker-run.mjs, which held the same constant and made the same claim; issue #227 moved the container-side
41
+ // half out, and this points at the half it actually wants -- a container path is not a Docker fact, and
42
+ // docker-run.mjs now has an import edge of its own.
43
+ import { CONTAINER_GLOBAL_PI_DIR } from "./container-spec.mjs";
41
44
 
42
45
  /** Staged packages live under `<globalPiDir>/packages/` -- a subdir of the overlay, not a new mount. */
43
46
  export const PACKAGES_SUBDIR = "packages";
package/src/processor.mjs CHANGED
@@ -1,3 +1,4 @@
1
+ import { DEFAULT_BACKEND, DOCKER_NEVER_STARTED_EXITS } from "./backends.mjs";
1
2
  import { lstatSync } from "node:fs";
2
3
  import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
3
4
  import { configError } from "./config.mjs";
@@ -100,6 +101,19 @@ export async function runJob(job, deps) {
100
101
  // would leave an INJECTED resolver running on every job, which is how a probe nobody wanted starts
101
102
  // spawning a subprocess per delivery to learn nothing.
102
103
  resolveSecrets = async (job) => ({ profileUnknown: job.secretsProfile ?? DEFAULT_SECRETS_PROFILE }),
104
+ // #227. Which backends this deployment BLESSED (PI_BACKENDS). Defaults to the one name every
105
+ // deployment already runs rather than to admit-everything: a wiring that says nothing blesses
106
+ // `local` only, so a job naming anything else is refused instead of running somewhere the operator
107
+ // never approved. That is the same fail-closed direction `resolveSecrets` above defaults in, and it
108
+ // is safe to default at all only because the gate below fires ONLY when a job names a backend --
109
+ // an unflagged job never consults this list.
110
+ blessedBackends = [DEFAULT_BACKEND],
111
+ // #227. The exit codes THIS JOB'S venue uses for "the runner never ran" -- a function of the job,
112
+ // not of the wiring, because which venue ran it is a per-job fact and the registry resolves it per
113
+ // job for every other backend function too. Docker's triple is the default because it is the only
114
+ // runtime this repo ships, so a wiring that omits this keeps today's behaviour exactly; an adapter
115
+ // that normalises to `container-never-started` itself returns an empty list.
116
+ neverStartedExits = () => DOCKER_NEVER_STARTED_EXITS,
103
117
  // (job) => scoped short-lived token. Takes the JOB, not the repo: which forge mints -- and therefore
104
118
  // which credential the container gets -- is a property of `job.kind`, and only the wiring knows the
105
119
  // map. Called for forge-backed jobs and for local jobs opted in via `github: true`; unflagged local
@@ -180,6 +194,25 @@ export async function runJob(job, deps) {
180
194
  }
181
195
  }
182
196
 
197
+ // #227. WHERE this job wants to run, against what this deployment blessed. FREE, determinate and
198
+ // credential-less, so it precedes the image inspect below for that gate's own stated reason: a job
199
+ // that names a venue this host will not use must refuse before anything spawns, mints or clones.
200
+ //
201
+ // The LOADER already refused a name this build does not know; what it could not check is PI_BACKENDS,
202
+ // which is a per-host setting a reviewed file must not be refused over -- the same split
203
+ // `run.secretsProfile` draws between its charset check at load and `secret-profile-unknown` here.
204
+ //
205
+ // Enforced HERE and not only in the panel's picker, because `DES-PER-TRIGGER-SECRET-PROFILE` says the
206
+ // overlay is not the reviewed artifact: a tool-side allowlist bounds what an operator can pick, and
207
+ // this bounds what actually runs.
208
+ if (job.backend !== undefined && !blessedBackends.includes(job.backend)) {
209
+ await comment(job, `Refused: this trigger asks to run on the "${job.backend}" backend, which this deployment does not bless. Not run.`);
210
+ // The backend NAME is operator-authored config, never payload, so naming it is PII-safe -- the
211
+ // same class as the image ref below.
212
+ log("refused_backend_unblessed", { backend: job.backend, blessed: blessedBackends });
213
+ return { outcome: "policy", reason: "backend-unblessed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
214
+ }
215
+
183
216
  // The job image must exist on THIS host before anything else happens. Free, determinate and
184
217
  // credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
185
218
  // image refuses without minting a credential it will not use, cloning a repo it will not read, or
@@ -611,18 +644,22 @@ export async function runJob(job, deps) {
611
644
  return { outcome: "policy", reason: "runner-policy", exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
612
645
  case EXIT_INFRA:
613
646
  throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
614
- case 125: // `docker run` itself failed (unusable image reference, bad flag)
615
- case 126: // the entrypoint exists but is not executable
616
- case 127: // the entrypoint was not found
617
- // In all three docker never handed control to the runner, so NOTHING was spent -- which is
618
- // exactly what `container-never-started` means, and it reuses the refund below rather than
619
- // keeping a slot the agent never used. These used to fall to `default:`, which kept the slot
620
- // AND retried, burning a second one. The preflight above converts the KNOWABLE case (an absent
621
- // image) into a pre-spend policy refusal; a 125 that survives it is a race (the image was
622
- // removed between the inspect and the run) or a docker-side fault we did not foresee --
623
- // genuinely infra, and now with a retry that costs nothing.
624
- throw new InfraRetry(`docker could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
625
647
  default:
648
+ // THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
649
+ // (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
650
+ // but is not executable) and 127 (the entrypoint was not found). Nothing was spent -- which is
651
+ // exactly what `container-never-started` means -- so this reuses the refund below rather than
652
+ // keeping a slot the agent never used. They used to fall to the unknown-exit branch, which
653
+ // kept the slot AND retried, burning a second one.
654
+ //
655
+ // ASKED OF THE BACKEND rather than hardcoded, because those integers are Docker's and they
656
+ // COLLIDE with the runner's own channel (`INT-RUNNER-EXIT-CODE-PROTOCOL`). Assuming them is
657
+ // silently wrong for any venue where 125 is a real runner exit, and the assumption was
658
+ // invisible while there was one runtime. An adapter declares its own set, or declares none
659
+ // and normalises to this outcome itself.
660
+ if ((neverStartedExits(job) ?? []).includes(code)) {
661
+ throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
662
+ }
626
663
  throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
627
664
  }
628
665
  } catch (e) {
package/src/queue.mjs CHANGED
@@ -78,7 +78,7 @@ export function makeQueue(connection, { name = QUEUE } = {}) {
78
78
  * removeOnComplete keeps the dedup window ~= the retention. Unlike webhooks, local jobs are not
79
79
  * redelivered, so a modest window is enough.
80
80
  */
81
- export async function enqueueLocalJob(queue, { folder, flow, task, command, provider, model, maxTurns, image, skillsDir, secrets, secretsProfile, chainDepth, parentJobId, jobId, now = new Date() }) {
81
+ export async function enqueueLocalJob(queue, { folder, flow, task, command, provider, model, maxTurns, image, backend, skillsDir, secrets, secretsProfile, chainDepth, parentJobId, jobId, now = new Date() }) {
82
82
  const minute = now.toISOString().slice(0, 16); // YYYY-MM-DDTHH:MM -- the dedup window
83
83
  // A caller-supplied jobId (the outbox collector's retry-idempotent chainedJobId) wins; otherwise the
84
84
  // minute-windowed localJobId is the dedup key. A command job (issue #189) fills the flow slot with
@@ -104,6 +104,12 @@ export async function enqueueLocalJob(queue, { folder, flow, task, command, prov
104
104
  model,
105
105
  maxTurns,
106
106
  ...(image !== undefined && { image }),
107
+ // #227. WHERE this job's container is built. Conditional like `image`, so a trigger that named no
108
+ // venue produces byte-identical job data -- and at JOB level, never inside `trigger`, for the reason
109
+ // `image` and `skillsDir` are: `trigger` is copied VERBATIM into `/job/event.json`, so a key added
110
+ // there becomes agent-visible input, and where the box was built is the worker's business rather
111
+ // than the agent's.
112
+ ...(backend !== undefined && { backend }),
107
113
  // The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
108
114
  // Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
109
115
  // than inside `trigger` because a worker-host path is an execution knob, not a fact about the
@@ -191,7 +197,7 @@ export async function enqueueGitLabJob(queue, fields) {
191
197
  * window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
192
198
  * has always been.
193
199
  */
194
- export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
200
+ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, backend, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
195
201
  const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
196
202
  // `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
197
203
  // come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
@@ -219,6 +225,12 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
219
225
  maxTurns,
220
226
  ...(packages !== undefined && { packages }),
221
227
  ...(image !== undefined && { image }),
228
+ // #227. WHERE this job's container is built. Conditional like `image`, so a trigger that named no
229
+ // venue produces byte-identical job data -- and at JOB level, never inside `trigger`, for the reason
230
+ // `image` and `skillsDir` are: `trigger` is copied VERBATIM into `/job/event.json`, so a key added
231
+ // there becomes agent-visible input, and where the box was built is the worker's business rather
232
+ // than the agent's.
233
+ ...(backend !== undefined && { backend }),
222
234
  // The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
223
235
  // Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
224
236
  // than inside `trigger` because a worker-host path is an execution knob, not a fact about the
@@ -1,11 +1,30 @@
1
1
  import { spawn } from "node:child_process";
2
2
  import { parseArgs } from "node:util";
3
+ import { DEFAULT_BACKEND, backendFor } from "./backends.mjs";
3
4
  import { loadConfig } from "./config.mjs";
4
5
  import { sanitizeJobId } from "./run-history.mjs";
5
6
  import { createJobNetwork, egressEnv, networkNameFor, removeJobNetwork } from "./egress.mjs";
6
7
  import { buildSandboxRunArgs, launchSandbox, listRunningSandboxes, parsePublish, resolveSandbox, sandboxContainerName } from "./sandbox.mjs";
7
8
  import { listSandboxes, pinSandbox } from "./sandbox-store.mjs";
8
9
 
10
+ /**
11
+ * Would this deployment's venues put a job's retained directory out of this command's reach? Exported ONLY
12
+ * so it can be driven.
13
+ *
14
+ * The refusal it answers is unreachable today: the table holds one backend and it is local, so
15
+ * `PI_BACKENDS` cannot name a remote venue. An unreachable rule with no test is a rule a mutation pass
16
+ * deletes in silence -- which is precisely what happened to the trigger-side remote refusal one slice ago,
17
+ * so this one is a pure predicate over an explicit config from the start.
18
+ *
19
+ * TRUE when any blessed venue is remote, not merely when the DEFAULT is: a job that ran there has a
20
+ * retained directory this host never had, and `pi-dispatch sandbox <jobId>` takes a job id rather than a
21
+ * venue, so it cannot know which one it is being asked about until the directory is already missing.
22
+ */
23
+ export function sandboxUnreachableFrom(config) {
24
+ const names = config?.backends ?? [config?.defaultBackend ?? DEFAULT_BACKEND];
25
+ return names.some((n) => backendFor(n)?.remote !== false);
26
+ }
27
+
9
28
  /**
10
29
  * `pi-dispatch sandbox` -- re-open a finished run's sandbox as an interactive shell
11
30
  * (REQ-RESURRECTABLE-SANDBOX).
@@ -78,6 +97,23 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
78
97
  return fail(err, error.message);
79
98
  }
80
99
 
100
+ // #227. THE SANDBOX IS LOCAL-ONLY, and this refusal is what keeps that true rather than accidental.
101
+ //
102
+ // `buildSandboxRunArgs` is a SECOND container producer, outside the `runContainer` seam and hard-wired to
103
+ // this host's docker CLI. It reopens a retained job directory: `manifest.workspace` is a path on THIS
104
+ // machine, which for a job that ran in another venue either does not exist or exists and reproduces a
105
+ // run from the wrong host, silently. `INT-SANDBOX-CONTRACT` is already "a SIBLING ... never an
106
+ // amendment" of the container contract, and this is the clause that says which sibling.
107
+ //
108
+ // Refused BEFORE `resolveSandbox` reads anything, so a deployment that blesses a remote venue is told
109
+ // what is wrong rather than handed a confusing missing-directory message.
110
+ if (sandboxUnreachableFrom(config)) {
111
+ return fail(
112
+ err,
113
+ `pi-dispatch sandbox opens a shell on THIS host's docker daemon against the job's retained directory, so it cannot reach a job that ran in a remote venue (PI_BACKENDS: ${config.backends.join(", ")}). Run it on the host that ran the job.`,
114
+ );
115
+ }
116
+
81
117
  const resolved = resolveSandbox({
82
118
  jobId,
83
119
  sandboxDir: config.sandboxDir,
package/src/schedules.mjs CHANGED
@@ -137,7 +137,7 @@ function normalizeCronSchedule({ on, run }, path, existsSync, fleet) {
137
137
  // key. A command trigger carries no flow/task at all (the validator enforces the XOR), so those two
138
138
  // keys hold undefined here and drop at JSON serialization -- the command schedule's data is exactly
139
139
  // kind/folder/command plus the shared fields.
140
- const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
140
+ const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.backend !== undefined && { backend: run.backend }), ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
141
141
  // Retention only; the deterministic repeat:<id>:<millis> jobId supplies dedup, so no jobId here, and
142
142
  // scheduler jobs are not retried (DES-CRON-VIA-BULLMQ-SCHEDULER) so no attempts/backoff.
143
143
  const opts = { removeOnComplete: { age: 24 * 3600 }, removeOnFail: { age: 7 * 24 * 3600 } };