@edgehero/pi-dispatch 1.9.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +9 -0
- package/package.json +7 -2
- package/src/backend-conformance.mjs +262 -0
- package/src/backend-local.mjs +222 -0
- package/src/backend-registry.mjs +157 -0
- package/src/backends.mjs +595 -0
- package/src/config.mjs +63 -1
- package/src/container-spec.mjs +226 -0
- package/src/docker-run.mjs +100 -106
- package/src/doctor.mjs +118 -0
- package/src/index.mjs +123 -10
- package/src/outbox.mjs +8 -0
- package/src/packages.mjs +7 -4
- package/src/processor.mjs +48 -11
- package/src/queue.mjs +14 -2
- package/src/sandbox-cli.mjs +36 -0
- package/src/schedules.mjs +1 -1
- package/src/start.mjs +118 -80
- package/src/triggers.mjs +120 -5
package/src/doctor.mjs
CHANGED
|
@@ -58,6 +58,7 @@ import { agentDirFrom, readHostPi } from "./host-pi.mjs";
|
|
|
58
58
|
import { PACKAGES_SUBDIR, readStagedSkills, readStageManifest } from "./packages.mjs";
|
|
59
59
|
import { copySkillTree } from "./copy-tree.mjs";
|
|
60
60
|
import { SKILL_NAME_RE } from "./flow-gate.mjs";
|
|
61
|
+
import { ABSENT, ASSERTED, PROPERTY_NAMES, declarationOf, floorShortfall, parseBackendFloor, parseBackendList, unarmedFloor } from "./backends.mjs";
|
|
61
62
|
import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
|
|
62
63
|
import { installedUnitPaths, readUnitSeam } from "./service.mjs";
|
|
63
64
|
import { parseSecretProfiles } from "./secret-profiles.mjs";
|
|
@@ -469,6 +470,7 @@ export async function collectChecks(env, seams) {
|
|
|
469
470
|
// byte-identical output. Gated on docker and the image, because two of these checks run a container and
|
|
470
471
|
// the rest are noise on top of a down daemon.
|
|
471
472
|
checks.push(...(await egressChecks(env, seams, { dockerCode, imageCode, jobImage })));
|
|
473
|
+
checks.push(...backendChecks(env));
|
|
472
474
|
|
|
473
475
|
// The receiver itself, when the triggers file names ANY forge (issue #80). Only forge deliveries need
|
|
474
476
|
// the receiver at all, so a cron/local-only deployment gets no receiver noise here. WARNS rather than
|
|
@@ -2337,3 +2339,119 @@ async function defaultProbeValkey(url) {
|
|
|
2337
2339
|
client.disconnect();
|
|
2338
2340
|
}
|
|
2339
2341
|
}
|
|
2342
|
+
|
|
2343
|
+
/**
|
|
2344
|
+
* WHERE this deployment's jobs run, and what that place actually guarantees (issue #227).
|
|
2345
|
+
*
|
|
2346
|
+
* THIS IS THE CHECK THAT MAKES THE DECLARATION ADMISSIBLE AT ALL. `CONST-EGRESS-POLICY-IN-THE-ARGV` says a
|
|
2347
|
+
* control an operator BELIEVES in is worse than one they know is missing, because the belief displaces the
|
|
2348
|
+
* credential bound that is really holding. A table of guarantees nothing ever prints is exactly such a
|
|
2349
|
+
* belief. So the three words must stay TOLD APART on the way out, and told apart ON THE SCREEN rather than
|
|
2350
|
+
* in a field nobody renders:
|
|
2351
|
+
*
|
|
2352
|
+
* enforced -- ours, in this worker's own code, readable back from what it produced. Quiet.
|
|
2353
|
+
* asserted -- someone else's. Rendered as a WARNING, and it NAMES who is asserting it, because "not us"
|
|
2354
|
+
* without "them" leaves an operator nothing to go and check.
|
|
2355
|
+
* absent -- not provided at all. A failure, since a deployment reaching it must know before a job does.
|
|
2356
|
+
*
|
|
2357
|
+
* `ok: false, warn: true` IS THE WARNING SHAPE, and it is the one thing to get right when editing here.
|
|
2358
|
+
* `render` reads `c.ok` FIRST, so `ok: true, warn: true` renders as a plain pass and drops the `fix` line
|
|
2359
|
+
* with it. An earlier draft used that shape and every asserted property printed as a green tick, which made
|
|
2360
|
+
* this section say the opposite of what it exists to say. `warn` keeps the RUN green -- `render` only fails
|
|
2361
|
+
* on `!ok && !warn` -- so an operator's CI is unaffected while the operator is actually told.
|
|
2362
|
+
*
|
|
2363
|
+
* A property a deployment switch gates is printed with the switch AND its position, never the bare
|
|
2364
|
+
* capability word: `local` can enforce egress, and a `PI_EGRESS=0` deployment is not getting it. Those are
|
|
2365
|
+
* two different sentences. `absent` OUTRANKS the gate, because a control that does not exist is a different
|
|
2366
|
+
* fact from one that is merely unarmed, and "CAN be absent but the switch is off" would be both meaningless
|
|
2367
|
+
* and green.
|
|
2368
|
+
*
|
|
2369
|
+
* Reads the environment directly, like every other check here, and parses through `backends.mjs` so doctor
|
|
2370
|
+
* and the worker cannot disagree about what a floor says.
|
|
2371
|
+
*/
|
|
2372
|
+
export function backendChecks(env) {
|
|
2373
|
+
const checks = [];
|
|
2374
|
+
let backends;
|
|
2375
|
+
let floor;
|
|
2376
|
+
try {
|
|
2377
|
+
backends = parseBackendList(env.PI_BACKENDS);
|
|
2378
|
+
floor = parseBackendFloor(env.PI_BACKEND_FLOOR);
|
|
2379
|
+
} catch (error) {
|
|
2380
|
+
// The worker refuses to boot on this, so doctor must not soften it to a warning.
|
|
2381
|
+
return [{ ok: false, label: `backend configuration does not parse: ${error.message}`, fix: "fix PI_BACKENDS / PI_BACKEND_FLOOR, then re-run doctor" }];
|
|
2382
|
+
}
|
|
2383
|
+
|
|
2384
|
+
// The switch positions every `armedBy` in the table can name. A MAP rather than one boolean, because
|
|
2385
|
+
// `armedBy` is a general field: hardcoding one variable name here would silently hide a second switch's
|
|
2386
|
+
// off-position the day one is added, which is the defect `armedBy` exists to prevent.
|
|
2387
|
+
const switches = {};
|
|
2388
|
+
try {
|
|
2389
|
+
switches.PI_EGRESS = egressArmed(env);
|
|
2390
|
+
} catch (error) {
|
|
2391
|
+
// NOT an abstention that falls through to the good case. A value doctor cannot parse is a value the
|
|
2392
|
+
// worker refuses to boot on, and an earlier draft claimed in a comment that "its own check reports
|
|
2393
|
+
// that" -- nothing did, so doctor printed every gated property as quietly enforced on a deployment
|
|
2394
|
+
// that could not start.
|
|
2395
|
+
checks.push({ ok: false, label: `PI_EGRESS does not parse, so what this deployment actually gets cannot be determined: ${error.message}`, fix: 'set PI_EGRESS to exactly "0" (off) or "1"/unset (on)' });
|
|
2396
|
+
}
|
|
2397
|
+
|
|
2398
|
+
// The dispatch landed in slice 4, so the "nothing selects yet" qualifier came off -- and it came off HERE
|
|
2399
|
+
// as well as in the code, because a stale caveat on the one surface that makes the table admissible is
|
|
2400
|
+
// its own kind of false statement.
|
|
2401
|
+
checks.push({ ok: true, label: `Jobs run on: ${backends.join(", ")}${backends.length > 1 ? ` (a trigger that names none runs on ${backends[0]}; run.backend selects)` : ""}` });
|
|
2402
|
+
|
|
2403
|
+
for (const name of backends) {
|
|
2404
|
+
for (const property of PROPERTY_NAMES) {
|
|
2405
|
+
const d = declarationOf(name, property);
|
|
2406
|
+
if (!d) continue;
|
|
2407
|
+
// FIRST, ahead of the gate: a control that does not exist is not a control that is unarmed.
|
|
2408
|
+
if (d.word === ABSENT) {
|
|
2409
|
+
checks.push({ ok: false, label: `${name}: ${property} is ABSENT -- ${d.question}`, fix: `this backend does not provide ${property}; a deployment that needs it must not run jobs on ${name}` });
|
|
2410
|
+
continue;
|
|
2411
|
+
}
|
|
2412
|
+
if (d.armedBy && switches[d.armedBy] === undefined) {
|
|
2413
|
+
// The switch did not parse. Say so rather than pick a side; the failure is already reported.
|
|
2414
|
+
checks.push({ ok: false, warn: true, label: `${name}: ${property} depends on ${d.armedBy}, which does not parse -- cannot say whether this deployment gets it`, fix: `fix ${d.armedBy}, then re-run doctor` });
|
|
2415
|
+
continue;
|
|
2416
|
+
}
|
|
2417
|
+
if (d.armedBy && switches[d.armedBy] === false) {
|
|
2418
|
+
checks.push({ ok: false, warn: true, label: `${name}: ${property} CAN be ${d.word} here, but ${d.armedBy} is off, so this deployment is not getting it`, fix: `arm ${d.armedBy} to get it (${d.question})` });
|
|
2419
|
+
continue;
|
|
2420
|
+
}
|
|
2421
|
+
if (d.word === ASSERTED) {
|
|
2422
|
+
checks.push({ ok: false, warn: true, label: `${name}: ${property} is ASSERTED by ${d.assertedBy ?? "something outside this worker"}, not enforced by it`, fix: `not verifiable from here, so treat it as a claim rather than a control: ${d.question}` });
|
|
2423
|
+
continue;
|
|
2424
|
+
}
|
|
2425
|
+
// enforced, and armed if it is gated at all. The good case, and it stays quiet.
|
|
2426
|
+
}
|
|
2427
|
+
}
|
|
2428
|
+
|
|
2429
|
+
// ALWAYS a line, including when no floor is set. `PI_BACKENDS_FLOOR` is a plausible one-character-off
|
|
2430
|
+
// spelling of the real name, and nothing in this project warns on an unknown PI_* variable, so silence
|
|
2431
|
+
// here would make a typo'd VARIABLE NAME look exactly like a floor that holds -- the same belief the
|
|
2432
|
+
// strict parsing inside the string exists to prevent, arriving from outside the string.
|
|
2433
|
+
const floorNames = Object.keys(floor);
|
|
2434
|
+
if (floorNames.length === 0) {
|
|
2435
|
+
checks.push({ ok: true, label: "PI_BACKEND_FLOOR is not set, so no minimum is required of any backend" });
|
|
2436
|
+
return checks;
|
|
2437
|
+
}
|
|
2438
|
+
|
|
2439
|
+
const misses = floorShortfall(backends, floor);
|
|
2440
|
+
const unarmed = unarmedFloor(floor, switches);
|
|
2441
|
+
// A floor whose every entry is `absent` parses, reads, and bounds NOTHING: `meets(have, absent)` is true
|
|
2442
|
+
// for every value. It is the one READABLE word that reproduces the outcome `isDeclaration` refuses a
|
|
2443
|
+
// typo for, so it is named rather than affirmed.
|
|
2444
|
+
const bounding = floorNames.filter((p) => floor[p] !== ABSENT);
|
|
2445
|
+
const spelled = floorNames.map((p) => `${p}=${floor[p]}`).join(", ");
|
|
2446
|
+
if (misses.length > 0) {
|
|
2447
|
+
checks.push({ ok: false, label: `PI_BACKEND_FLOOR is not met: ${misses.map((m) => `${m.backend}.${m.property} is ${m.have}`).join(", ")}`, fix: "raise the backend, lower PI_BACKEND_FLOOR, or drop the backend from PI_BACKENDS" });
|
|
2448
|
+
} else if (unarmed.length > 0) {
|
|
2449
|
+
checks.push({ ok: false, label: `PI_BACKEND_FLOOR asks for ${unarmed.map((u) => `${u.property}=${u.want}`).join(", ")}, which ${[...new Set(unarmed.map((u) => u.armedBy))].join(", ")} has switched off`, fix: "arm the switch, or lower that entry to `absent` if you did not mean to require it" });
|
|
2450
|
+
} else if (bounding.length === 0) {
|
|
2451
|
+
checks.push({ ok: false, warn: true, label: `PI_BACKEND_FLOOR (${spelled}) requires nothing: every entry asks for "absent", which every backend meets`, fix: "raise an entry to `asserted` or `enforced` for it to bound anything" });
|
|
2452
|
+
} else {
|
|
2453
|
+
checks.push({ ok: true, label: `PI_BACKEND_FLOOR holds (${spelled})` });
|
|
2454
|
+
}
|
|
2455
|
+
|
|
2456
|
+
return checks;
|
|
2457
|
+
}
|
package/src/index.mjs
CHANGED
|
@@ -1,13 +1,11 @@
|
|
|
1
|
-
import { execFile } from "node:child_process";
|
|
2
|
-
import { promisify } from "node:util";
|
|
3
1
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
2
|
+
import { jobContainerName } from "./backend-local.mjs";
|
|
4
3
|
import { InfraRetry, runJob } from "./processor.mjs";
|
|
5
4
|
import { targetFor } from "./run-history.mjs";
|
|
6
5
|
import { budgetCapsFor, canonicalScope, concurrencyFor, makeInFlight, scopeKeyPrefix } from "./scoped-limits.mjs";
|
|
7
6
|
import { WAIT_AFTER_MAX_DEFAULT_MS, WAIT_INTERVAL_FLOOR_MS, afterMs, unreadableConditions, waitArmed, waitBackoffMs, waitLabel, waitProfileNames } from "./wait-for.mjs";
|
|
8
7
|
import { makeWaitState } from "./wait-state.mjs";
|
|
9
8
|
|
|
10
|
-
const exec = promisify(execFile);
|
|
11
9
|
|
|
12
10
|
export const QUEUE = "pi-jobs";
|
|
13
11
|
|
|
@@ -44,6 +42,50 @@ const THROTTLE_ALARM = 5;
|
|
|
44
42
|
// the only evidence there is. A test pins the two apart.
|
|
45
43
|
const THROTTLE_FLOOR_MS = 11_000;
|
|
46
44
|
|
|
45
|
+
/**
|
|
46
|
+
* How long a container gets to actually die after the abort's `docker stop` before the worker stops waiting.
|
|
47
|
+
*
|
|
48
|
+
* `docker stop -t 5` is SIGTERM then an unignorable SIGKILL five seconds later, so a reachable daemon ends
|
|
49
|
+
* the container well inside this. The margin is for the daemon being slow, not for the container being
|
|
50
|
+
* stubborn -- a container cannot outlive SIGKILL.
|
|
51
|
+
*/
|
|
52
|
+
const ABORT_GRACE_MS = 30_000;
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Resolve `run` normally, but stop waiting once the abort has fired and the grace has passed.
|
|
56
|
+
*
|
|
57
|
+
* See the call site for why this exists. Returns the same `{ code: 137, aborted: true }` shape a killed
|
|
58
|
+
* container produces, so nothing downstream needs to know the difference -- the processor's abort
|
|
59
|
+
* classification, the run record and the refund all behave exactly as they do for a stop that worked.
|
|
60
|
+
*/
|
|
61
|
+
function boundAfterAbort(run, signal, job, log, graceMs = ABORT_GRACE_MS) {
|
|
62
|
+
if (!signal) return run;
|
|
63
|
+
return new Promise((resolve, reject) => {
|
|
64
|
+
let timer = null;
|
|
65
|
+
let settled = false;
|
|
66
|
+
const done = (fn) => (v) => {
|
|
67
|
+
if (settled) return;
|
|
68
|
+
settled = true;
|
|
69
|
+
clearTimeout(timer);
|
|
70
|
+
fn(v);
|
|
71
|
+
};
|
|
72
|
+
const onAbort = () => {
|
|
73
|
+
timer = setTimeout(() => {
|
|
74
|
+
if (settled) return;
|
|
75
|
+
settled = true;
|
|
76
|
+
log("stop_did_not_take", { job: job.id, graceMs });
|
|
77
|
+
resolve({ code: 137, aborted: true, turns: null, tokens: null, session: null, usage: null, context: null });
|
|
78
|
+
}, graceMs);
|
|
79
|
+
// A boot-blocking handle is not wanted here: the worker should be able to exit if everything else
|
|
80
|
+
// has finished, and this timer only matters while a job is still in flight.
|
|
81
|
+
timer.unref?.();
|
|
82
|
+
};
|
|
83
|
+
if (signal.aborted) onAbort();
|
|
84
|
+
else signal.addEventListener("abort", onAbort, { once: true });
|
|
85
|
+
run.then(done(resolve), done(reject));
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
|
|
47
89
|
/**
|
|
48
90
|
* Build the BullMQ processor.
|
|
49
91
|
*
|
|
@@ -53,7 +95,8 @@ const THROTTLE_FLOOR_MS = 11_000;
|
|
|
53
95
|
* abort -- with no error. A test asserts the arity precisely because the failure is silent.
|
|
54
96
|
*
|
|
55
97
|
* Dependencies are injected so this is testable without a live queue: `cancelJob` (fired by the
|
|
56
|
-
* timeout), `stopContainer` (fired by the abort
|
|
98
|
+
* timeout), `stopContainer` (fired by the abort, and INJECTED so the venue that built the container is
|
|
99
|
+
* the one that stops it), and the orchestration deps.
|
|
57
100
|
*
|
|
58
101
|
* Once per job, before runJob, it resolves the runtime-settings overlay via `getSettings`
|
|
59
102
|
* (INT-CONFIG-OVERLAY-CONTRACT). A present-but-invalid overlay resolves to a POLICY refusal RETURNED
|
|
@@ -63,7 +106,7 @@ const THROTTLE_FLOOR_MS = 11_000;
|
|
|
63
106
|
* The overlay changes which values the spend caps take, never when they are checked -- reserveBudget still
|
|
64
107
|
* runs inside runJob against the freshly passed caps (CONST-BUDGET-BEFORE-TOKENS).
|
|
65
108
|
*/
|
|
66
|
-
export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
109
|
+
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, deps, recordRun = () => {}, timeoutMs = JOB_TIMEOUT_MS, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random }) {
|
|
67
110
|
return async function processor(job, token, signal) {
|
|
68
111
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
69
112
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
@@ -491,6 +534,7 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
491
534
|
let name;
|
|
492
535
|
let timer;
|
|
493
536
|
let onAbort;
|
|
537
|
+
let venue;
|
|
494
538
|
try {
|
|
495
539
|
// Nothing between the acquire above and the main `try` below may throw unguarded: the releasing
|
|
496
540
|
// finally belongs to THAT try, so an unguarded throw here would leak the hold and wedge the
|
|
@@ -498,7 +542,23 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
498
542
|
// addEventListener on the bullmq-allocated controller are total at processor arity 3); the
|
|
499
543
|
// guard is structural, not observational.
|
|
500
544
|
startedAt = new Date().toISOString();
|
|
501
|
-
name
|
|
545
|
+
// The producer of the name both boot reapers sweep by substring. Built from the shared prefix
|
|
546
|
+
// rather than typed here, so a rename cannot land in the producer and not in the sweeps (#227).
|
|
547
|
+
// From the VENUE that will build the container, not from the local adapter reached for directly:
|
|
548
|
+
// the abort stops this name, so the name and the stop have to come from the same backend. The
|
|
549
|
+
// default keeps every wiring that predates the seam building it exactly as before.
|
|
550
|
+
//
|
|
551
|
+
// `job.data`, NOT `job`. This function's `job` is the BullMQ WRAPPER -- its own keys are `id` and
|
|
552
|
+
// `data` -- so `job.backend` is always undefined and the registry's resolution would fall to
|
|
553
|
+
// `?? defaultName` for every job, silently, on the one path where the fail-closed throw can never
|
|
554
|
+
// fire because "names nothing" is exactly the case it permits. `runJob` is handed `effectiveJob`,
|
|
555
|
+
// a spread of `job.data`, which is why `runContainer` and the two preflights dispatch correctly
|
|
556
|
+
// while these two did not.
|
|
557
|
+
// One object carrying BOTH halves: the id is the BullMQ wrapper's (it always was) and the venue is
|
|
558
|
+
// the trigger's, which lives in `data`. Built once so the name and the stop cannot resolve
|
|
559
|
+
// different backends -- the whole reason the name moved onto the registry in the first place.
|
|
560
|
+
venue = { ...job.data, id: job.id };
|
|
561
|
+
name = containerName(venue);
|
|
502
562
|
timer = setTimeout(() => {
|
|
503
563
|
// BullMQ has no per-job kill timer; this is ours. cancelJob raises the AbortSignal.
|
|
504
564
|
Promise.resolve(cancelJob(job.id, "job-timeout-30m")).catch(() => {});
|
|
@@ -507,7 +567,23 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
507
567
|
// Abort (timeout OR shutdown) => stop the container. docker stop sends SIGTERM then SIGKILL
|
|
508
568
|
// after the grace period; the runner exits and runContainer returns/throws.
|
|
509
569
|
onAbort = () => {
|
|
510
|
-
|
|
570
|
+
// The JOB'S DATA goes with the name -- `job.data`, not `job`. A container name alone cannot say
|
|
571
|
+
// which runtime holds it once there is more than one venue, and this call is the only thing
|
|
572
|
+
// standing between a runaway job and the 30-minute bound (REQ-JOB-TIMEOUT-30M). Passing the
|
|
573
|
+
// BullMQ wrapper sent every abort to the DEFAULT venue: `docker stop` on a host that never
|
|
574
|
+
// held the container, rejecting into a log line while the real one kept running and kept
|
|
575
|
+
// spending, with the local reaper unable to see it either.
|
|
576
|
+
// The whole call is inside the try, not just its promise. `Promise.resolve(f())` evaluates `f()`
|
|
577
|
+
// FIRST, so a missing or throwing `stopContainer` raises synchronously, inside an
|
|
578
|
+
// AbortSignal listener, where it surfaces as an uncaughtException and takes the worker
|
|
579
|
+
// process down 30 minutes into a runaway job -- killing every other in-flight job on the
|
|
580
|
+
// host. Losing the kill for one job is bad; losing the process is worse.
|
|
581
|
+
const note = deps.log ?? (() => {});
|
|
582
|
+
try {
|
|
583
|
+
Promise.resolve(stopContainer(name, venue)).catch((err) => note("stop_container_failed", { job: job.id, reason: err?.message }));
|
|
584
|
+
} catch (err) {
|
|
585
|
+
note("stop_container_failed", { job: job.id, reason: err?.message });
|
|
586
|
+
}
|
|
511
587
|
};
|
|
512
588
|
signal.addEventListener("abort", onAbort, { once: true });
|
|
513
589
|
} catch (error) {
|
|
@@ -574,7 +650,26 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
574
650
|
// life. Null when no row carries a money window for this scope.
|
|
575
651
|
scopedCaps: budgetCapsFor(job.data, limits),
|
|
576
652
|
...deps,
|
|
577
|
-
|
|
653
|
+
// #227. BOUNDED AFTER THE ABORT, and this is what makes `abortable` an honest declaration.
|
|
654
|
+
//
|
|
655
|
+
// `makeRunContainer`'s promise settles ONLY on the docker child's `close` or `error`. Nothing
|
|
656
|
+
// else ends that await -- the signal is read at entry and captured at close, never passed to
|
|
657
|
+
// the spawn. So `stopContainer` is the sole kill channel, and if it does not take (an
|
|
658
|
+
// unreachable daemon, a wiring whose stop is a no-op) the container keeps running, `docker
|
|
659
|
+
// run` never exits, and the processor awaits FOREVER: an active job renewing its lock and
|
|
660
|
+
// holding its in-flight slot, its host slot, its scope lease and its budget reservation, with
|
|
661
|
+
// nothing in the log and nothing in the record. REQ-JOB-TIMEOUT-30M's Acceptance says "the
|
|
662
|
+
// slot is freed", and it was not.
|
|
663
|
+
//
|
|
664
|
+
// So once the abort has fired, the wait is bounded. On expiry this resolves the SAME shape a
|
|
665
|
+
// killed container returns -- `{ code: 137, aborted: true }`, which `run-container.mjs`
|
|
666
|
+
// already uses for "aborted before it could start" -- so the processor classifies it as
|
|
667
|
+
// POLICY and does not retry. That is deliberate: the container may still be running, and a
|
|
668
|
+
// retry would pay for a second one alongside it. What is leaked is the container, which the
|
|
669
|
+
// next boot reaper sweeps; what is NOT leaked is the slot, the lease and the reservation.
|
|
670
|
+
// The `stop_did_not_take` line is the loud half, because a host whose daemon ignores a stop
|
|
671
|
+
// is a fact an operator has to learn from somewhere.
|
|
672
|
+
runContainer: (ctx) => boundAfterAbort(deps.runContainer({ ...ctx, name, signal }), signal, job, deps.log ?? (() => {})),
|
|
578
673
|
// REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
|
|
579
674
|
// to be abortable for the same reason runContainer does: a resolver blocking on an unreachable
|
|
580
675
|
// vault would otherwise hold its slot until its own timeout, and an abort landing mid-resolution
|
|
@@ -630,12 +725,19 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
630
725
|
};
|
|
631
726
|
}
|
|
632
727
|
|
|
633
|
-
export function createWorker({ connection, name, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
|
|
728
|
+
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, limiter, pauseUntil, scopedLimits, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), extraClosers = [] }) {
|
|
634
729
|
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
635
730
|
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
636
731
|
// because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
|
|
637
732
|
// the delayed set is gated on each worker's own `Date.now()`, so the fastest clock wins every hop and a
|
|
638
733
|
// job that had to reach another host might never get there.
|
|
734
|
+
// #227. REFUSED AT CONSTRUCTION, because the alternative is discovering it 30 minutes into a runaway
|
|
735
|
+
// job. `stopContainer` is the only thing that enforces REQ-JOB-TIMEOUT-30M, and it is what the backend
|
|
736
|
+
// table declares as `abortable: enforced` -- a wiring that omits it would make that declaration false
|
|
737
|
+
// while every test that never aborts stayed green.
|
|
738
|
+
if (typeof stopContainer !== "function") {
|
|
739
|
+
throw new Error("createWorker: stopContainer is required -- it is the only thing that enforces the 30-minute job timeout");
|
|
740
|
+
}
|
|
639
741
|
const names = hostQueue ? [QUEUE, hostQueue] : [QUEUE];
|
|
640
742
|
const workers = [];
|
|
641
743
|
|
|
@@ -655,7 +757,12 @@ export function createWorker({ connection, name, hostQueue = null, checkLease =
|
|
|
655
757
|
// Bound to THIS worker: a job on the host queue is cancelled by the worker draining that queue,
|
|
656
758
|
// and the shared handle could not reach it.
|
|
657
759
|
cancelJob: (id, reason) => worker.cancelJob(id, reason),
|
|
658
|
-
|
|
760
|
+
// #227. INJECTED, not built here. This was a one-line `docker stop` literal, which meant the abort
|
|
761
|
+
// path -- the only thing that can end a runaway job -- was the one backend function unreachable
|
|
762
|
+
// from `startWorker`. The wiring now passes the registry's per-job stop, so the container is
|
|
763
|
+
// stopped by whatever venue built it.
|
|
764
|
+
stopContainer,
|
|
765
|
+
containerName,
|
|
659
766
|
redis,
|
|
660
767
|
getSettings,
|
|
661
768
|
// Late-bound over EVERY worker: an overlay concurrency change re-binds the live slot count at the
|
|
@@ -733,3 +840,9 @@ export function createWorker({ connection, name, hostQueue = null, checkLease =
|
|
|
733
840
|
|
|
734
841
|
return primary;
|
|
735
842
|
}
|
|
843
|
+
|
|
844
|
+
/**
|
|
845
|
+
* `boundAfterAbort` with a zero grace, for tests only. The real bound is 30 seconds, which no test can wait
|
|
846
|
+
* for, and the behaviour under test is what happens WHEN the grace expires -- not how long it is.
|
|
847
|
+
*/
|
|
848
|
+
export const __boundAfterAbortForTests = (run, signal, job, log) => boundAfterAbort(run, signal, job, log, 0);
|
package/src/outbox.mjs
CHANGED
|
@@ -171,6 +171,14 @@ export function makeCollectChain({ queue, enqueue = enqueueLocalJob, readFlowGat
|
|
|
171
171
|
// (INT-OUTBOX-CONTRACT's explicit-property-reads rule). Undefined stays undefined, so a parent with
|
|
172
172
|
// no image chains a child whose data is byte-identical to today's.
|
|
173
173
|
image: job.data?.image,
|
|
174
|
+
// #227, and INHERITED for the reason `image` directly above is: a chained child continues its
|
|
175
|
+
// parent's work, so it belongs in the venue the parent's trigger chose, not silently back on
|
|
176
|
+
// the deployment default. Written down rather than left to inference because this file's
|
|
177
|
+
// convention is that every inherit/don't-inherit decision here carries its reason -- and the
|
|
178
|
+
// distinction it turns on is the same one `secrets` is excluded by: a venue is toolchain, a
|
|
179
|
+
// resolved credential is a capability. The agent cannot choose it any more than it can choose
|
|
180
|
+
// its child's image: the value comes off `job.data`, never off the request file.
|
|
181
|
+
backend: job.data?.backend,
|
|
174
182
|
// The parent's injected skills follow the child, off validated JOB DATA and never off the
|
|
175
183
|
// request file (REQ-PER-TRIGGER-SKILLS). Same reason `image` does: a chained child runs the
|
|
176
184
|
// same operator's flows and, without them, would look up a skill that is not there, write a
|
package/src/packages.mjs
CHANGED
|
@@ -34,10 +34,13 @@ import { configError } from "./config.mjs";
|
|
|
34
34
|
// and the admin block cannot drift between the stager and this validator (doctor.mjs sets the precedent
|
|
35
35
|
// of importing from import-pi.mjs).
|
|
36
36
|
import { ADMIN_RE, ENTRY_NAME_RE } from "./import-pi.mjs";
|
|
37
|
-
// The container-side mount point is
|
|
38
|
-
// packages root below cannot drift apart while both test suites stay green.
|
|
39
|
-
//
|
|
40
|
-
|
|
37
|
+
// The container-side mount point is container-spec.mjs's fact -- IMPORTED, never re-typed, so the mount and
|
|
38
|
+
// the packages root below cannot drift apart while both test suites stay green. That module is a LEAF and
|
|
39
|
+
// imports nothing, so this costs no cycle and no weight in the admin's bundle. It used to be imported from
|
|
40
|
+
// docker-run.mjs, which held the same constant and made the same claim; issue #227 moved the container-side
|
|
41
|
+
// half out, and this points at the half it actually wants -- a container path is not a Docker fact, and
|
|
42
|
+
// docker-run.mjs now has an import edge of its own.
|
|
43
|
+
import { CONTAINER_GLOBAL_PI_DIR } from "./container-spec.mjs";
|
|
41
44
|
|
|
42
45
|
/** Staged packages live under `<globalPiDir>/packages/` -- a subdir of the overlay, not a new mount. */
|
|
43
46
|
export const PACKAGES_SUBDIR = "packages";
|
package/src/processor.mjs
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { DEFAULT_BACKEND, DOCKER_NEVER_STARTED_EXITS } from "./backends.mjs";
|
|
1
2
|
import { lstatSync } from "node:fs";
|
|
2
3
|
import { checkTokenCap, recordTokenSpend, releaseBudget, reserveBudget } from "./budget.mjs";
|
|
3
4
|
import { configError } from "./config.mjs";
|
|
@@ -100,6 +101,19 @@ export async function runJob(job, deps) {
|
|
|
100
101
|
// would leave an INJECTED resolver running on every job, which is how a probe nobody wanted starts
|
|
101
102
|
// spawning a subprocess per delivery to learn nothing.
|
|
102
103
|
resolveSecrets = async (job) => ({ profileUnknown: job.secretsProfile ?? DEFAULT_SECRETS_PROFILE }),
|
|
104
|
+
// #227. Which backends this deployment BLESSED (PI_BACKENDS). Defaults to the one name every
|
|
105
|
+
// deployment already runs rather than to admit-everything: a wiring that says nothing blesses
|
|
106
|
+
// `local` only, so a job naming anything else is refused instead of running somewhere the operator
|
|
107
|
+
// never approved. That is the same fail-closed direction `resolveSecrets` above defaults in, and it
|
|
108
|
+
// is safe to default at all only because the gate below fires ONLY when a job names a backend --
|
|
109
|
+
// an unflagged job never consults this list.
|
|
110
|
+
blessedBackends = [DEFAULT_BACKEND],
|
|
111
|
+
// #227. The exit codes THIS JOB'S venue uses for "the runner never ran" -- a function of the job,
|
|
112
|
+
// not of the wiring, because which venue ran it is a per-job fact and the registry resolves it per
|
|
113
|
+
// job for every other backend function too. Docker's triple is the default because it is the only
|
|
114
|
+
// runtime this repo ships, so a wiring that omits this keeps today's behaviour exactly; an adapter
|
|
115
|
+
// that normalises to `container-never-started` itself returns an empty list.
|
|
116
|
+
neverStartedExits = () => DOCKER_NEVER_STARTED_EXITS,
|
|
103
117
|
// (job) => scoped short-lived token. Takes the JOB, not the repo: which forge mints -- and therefore
|
|
104
118
|
// which credential the container gets -- is a property of `job.kind`, and only the wiring knows the
|
|
105
119
|
// map. Called for forge-backed jobs and for local jobs opted in via `github: true`; unflagged local
|
|
@@ -180,6 +194,25 @@ export async function runJob(job, deps) {
|
|
|
180
194
|
}
|
|
181
195
|
}
|
|
182
196
|
|
|
197
|
+
// #227. WHERE this job wants to run, against what this deployment blessed. FREE, determinate and
|
|
198
|
+
// credential-less, so it precedes the image inspect below for that gate's own stated reason: a job
|
|
199
|
+
// that names a venue this host will not use must refuse before anything spawns, mints or clones.
|
|
200
|
+
//
|
|
201
|
+
// The LOADER already refused a name this build does not know; what it could not check is PI_BACKENDS,
|
|
202
|
+
// which is a per-host setting a reviewed file must not be refused over -- the same split
|
|
203
|
+
// `run.secretsProfile` draws between its charset check at load and `secret-profile-unknown` here.
|
|
204
|
+
//
|
|
205
|
+
// Enforced HERE and not only in the panel's picker, because `DES-PER-TRIGGER-SECRET-PROFILE` says the
|
|
206
|
+
// overlay is not the reviewed artifact: a tool-side allowlist bounds what an operator can pick, and
|
|
207
|
+
// this bounds what actually runs.
|
|
208
|
+
if (job.backend !== undefined && !blessedBackends.includes(job.backend)) {
|
|
209
|
+
await comment(job, `Refused: this trigger asks to run on the "${job.backend}" backend, which this deployment does not bless. Not run.`);
|
|
210
|
+
// The backend NAME is operator-authored config, never payload, so naming it is PII-safe -- the
|
|
211
|
+
// same class as the image ref below.
|
|
212
|
+
log("refused_backend_unblessed", { backend: job.backend, blessed: blessedBackends });
|
|
213
|
+
return { outcome: "policy", reason: "backend-unblessed", exitCode: null, turns: null, tokens: null, provider: job.provider ?? null, model: job.model ?? null, budgetReserved: false }; // return => not retried
|
|
214
|
+
}
|
|
215
|
+
|
|
183
216
|
// The job image must exist on THIS host before anything else happens. Free, determinate and
|
|
184
217
|
// credential-less, so it precedes the mint, the clone and the reservation: a host that cannot run the
|
|
185
218
|
// image refuses without minting a credential it will not use, cloning a repo it will not read, or
|
|
@@ -611,18 +644,22 @@ export async function runJob(job, deps) {
|
|
|
611
644
|
return { outcome: "policy", reason: "runner-policy", exitCode: code, turns, tokens, usage: usage ?? null, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session), budgetReserved: true };
|
|
612
645
|
case EXIT_INFRA:
|
|
613
646
|
throw new InfraRetry(`infra failure, container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
614
|
-
case 125: // `docker run` itself failed (unusable image reference, bad flag)
|
|
615
|
-
case 126: // the entrypoint exists but is not executable
|
|
616
|
-
case 127: // the entrypoint was not found
|
|
617
|
-
// In all three docker never handed control to the runner, so NOTHING was spent -- which is
|
|
618
|
-
// exactly what `container-never-started` means, and it reuses the refund below rather than
|
|
619
|
-
// keeping a slot the agent never used. These used to fall to `default:`, which kept the slot
|
|
620
|
-
// AND retried, burning a second one. The preflight above converts the KNOWABLE case (an absent
|
|
621
|
-
// image) into a pre-spend policy refusal; a 125 that survives it is a race (the image was
|
|
622
|
-
// removed between the inspect and the run) or a docker-side fault we did not foresee --
|
|
623
|
-
// genuinely infra, and now with a retry that costs nothing.
|
|
624
|
-
throw new InfraRetry(`docker could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
625
647
|
default:
|
|
648
|
+
// THE RUNTIME NEVER HANDED CONTROL TO THE RUNNER, in whatever integers this venue spells that
|
|
649
|
+
// (issue #227). For docker it is 125 (`docker run` itself failed), 126 (the entrypoint exists
|
|
650
|
+
// but is not executable) and 127 (the entrypoint was not found). Nothing was spent -- which is
|
|
651
|
+
// exactly what `container-never-started` means -- so this reuses the refund below rather than
|
|
652
|
+
// keeping a slot the agent never used. They used to fall to the unknown-exit branch, which
|
|
653
|
+
// kept the slot AND retried, burning a second one.
|
|
654
|
+
//
|
|
655
|
+
// ASKED OF THE BACKEND rather than hardcoded, because those integers are Docker's and they
|
|
656
|
+
// COLLIDE with the runner's own channel (`INT-RUNNER-EXIT-CODE-PROTOCOL`). Assuming them is
|
|
657
|
+
// silently wrong for any venue where 125 is a real runner exit, and the assumption was
|
|
658
|
+
// invisible while there was one runtime. An adapter declares its own set, or declares none
|
|
659
|
+
// and normalises to this outcome itself.
|
|
660
|
+
if ((neverStartedExits(job) ?? []).includes(code)) {
|
|
661
|
+
throw new InfraRetry(`the runtime could not start the container, exit ${code}`, { reason: "container-never-started", exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
662
|
+
}
|
|
626
663
|
throw new InfraRetry(`unknown container exit ${code}`, { exitCode: code, turns, tokens, usage, provider: job.provider ?? null, model: job.model ?? null, session: mergeSession(prepared, session) });
|
|
627
664
|
}
|
|
628
665
|
} catch (e) {
|
package/src/queue.mjs
CHANGED
|
@@ -78,7 +78,7 @@ export function makeQueue(connection, { name = QUEUE } = {}) {
|
|
|
78
78
|
* removeOnComplete keeps the dedup window ~= the retention. Unlike webhooks, local jobs are not
|
|
79
79
|
* redelivered, so a modest window is enough.
|
|
80
80
|
*/
|
|
81
|
-
export async function enqueueLocalJob(queue, { folder, flow, task, command, provider, model, maxTurns, image, skillsDir, secrets, secretsProfile, chainDepth, parentJobId, jobId, now = new Date() }) {
|
|
81
|
+
export async function enqueueLocalJob(queue, { folder, flow, task, command, provider, model, maxTurns, image, backend, skillsDir, secrets, secretsProfile, chainDepth, parentJobId, jobId, now = new Date() }) {
|
|
82
82
|
const minute = now.toISOString().slice(0, 16); // YYYY-MM-DDTHH:MM -- the dedup window
|
|
83
83
|
// A caller-supplied jobId (the outbox collector's retry-idempotent chainedJobId) wins; otherwise the
|
|
84
84
|
// minute-windowed localJobId is the dedup key. A command job (issue #189) fills the flow slot with
|
|
@@ -104,6 +104,12 @@ export async function enqueueLocalJob(queue, { folder, flow, task, command, prov
|
|
|
104
104
|
model,
|
|
105
105
|
maxTurns,
|
|
106
106
|
...(image !== undefined && { image }),
|
|
107
|
+
// #227. WHERE this job's container is built. Conditional like `image`, so a trigger that named no
|
|
108
|
+
// venue produces byte-identical job data -- and at JOB level, never inside `trigger`, for the reason
|
|
109
|
+
// `image` and `skillsDir` are: `trigger` is copied VERBATIM into `/job/event.json`, so a key added
|
|
110
|
+
// there becomes agent-visible input, and where the box was built is the worker's business rather
|
|
111
|
+
// than the agent's.
|
|
112
|
+
...(backend !== undefined && { backend }),
|
|
107
113
|
// The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
|
|
108
114
|
// Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
|
|
109
115
|
// than inside `trigger` because a worker-host path is an execution knob, not a fact about the
|
|
@@ -191,7 +197,7 @@ export async function enqueueGitLabJob(queue, fields) {
|
|
|
191
197
|
* window, replicas never coalesce against each other, and an unflagged job's dedup id is the same string it
|
|
192
198
|
* has always been.
|
|
193
199
|
*/
|
|
194
|
-
export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
|
|
200
|
+
export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, target, flow, command, trigger, provider, model, maxTurns, packages, image, backend, skillsDir, instructions, resume, secrets, secretsProfile, waitFor, replica, replicas }) {
|
|
195
201
|
const jobId = forgeDeliveryJobId(kind, trigger?.deliveryId, replica);
|
|
196
202
|
// `packages` (whether to load the operator-staged pi packages) and `image` (which container image to run)
|
|
197
203
|
// come off the MATCHED trigger (INT-TRIGGERS-FILE-CONTRACT / REQ-GLOBAL-PI-OVERLAY) and land on `data`
|
|
@@ -219,6 +225,12 @@ export async function enqueueForgeJob(queue, kind, { repo, projectId, azure, tar
|
|
|
219
225
|
maxTurns,
|
|
220
226
|
...(packages !== undefined && { packages }),
|
|
221
227
|
...(image !== undefined && { image }),
|
|
228
|
+
// #227. WHERE this job's container is built. Conditional like `image`, so a trigger that named no
|
|
229
|
+
// venue produces byte-identical job data -- and at JOB level, never inside `trigger`, for the reason
|
|
230
|
+
// `image` and `skillsDir` are: `trigger` is copied VERBATIM into `/job/event.json`, so a key added
|
|
231
|
+
// there becomes agent-visible input, and where the box was built is the worker's business rather
|
|
232
|
+
// than the agent's.
|
|
233
|
+
...(backend !== undefined && { backend }),
|
|
222
234
|
// The host directory of operator-authored skills this trigger injects (REQ-PER-TRIGGER-SKILLS).
|
|
223
235
|
// Conditional like `image`, so an unflagged job's data stays byte-identical, and at JOB level rather
|
|
224
236
|
// than inside `trigger` because a worker-host path is an execution knob, not a fact about the
|
package/src/sandbox-cli.mjs
CHANGED
|
@@ -1,11 +1,30 @@
|
|
|
1
1
|
import { spawn } from "node:child_process";
|
|
2
2
|
import { parseArgs } from "node:util";
|
|
3
|
+
import { DEFAULT_BACKEND, backendFor } from "./backends.mjs";
|
|
3
4
|
import { loadConfig } from "./config.mjs";
|
|
4
5
|
import { sanitizeJobId } from "./run-history.mjs";
|
|
5
6
|
import { createJobNetwork, egressEnv, networkNameFor, removeJobNetwork } from "./egress.mjs";
|
|
6
7
|
import { buildSandboxRunArgs, launchSandbox, listRunningSandboxes, parsePublish, resolveSandbox, sandboxContainerName } from "./sandbox.mjs";
|
|
7
8
|
import { listSandboxes, pinSandbox } from "./sandbox-store.mjs";
|
|
8
9
|
|
|
10
|
+
/**
|
|
11
|
+
* Would this deployment's venues put a job's retained directory out of this command's reach? Exported ONLY
|
|
12
|
+
* so it can be driven.
|
|
13
|
+
*
|
|
14
|
+
* The refusal it answers is unreachable today: the table holds one backend and it is local, so
|
|
15
|
+
* `PI_BACKENDS` cannot name a remote venue. An unreachable rule with no test is a rule a mutation pass
|
|
16
|
+
* deletes in silence -- which is precisely what happened to the trigger-side remote refusal one slice ago,
|
|
17
|
+
* so this one is a pure predicate over an explicit config from the start.
|
|
18
|
+
*
|
|
19
|
+
* TRUE when any blessed venue is remote, not merely when the DEFAULT is: a job that ran there has a
|
|
20
|
+
* retained directory this host never had, and `pi-dispatch sandbox <jobId>` takes a job id rather than a
|
|
21
|
+
* venue, so it cannot know which one it is being asked about until the directory is already missing.
|
|
22
|
+
*/
|
|
23
|
+
export function sandboxUnreachableFrom(config) {
|
|
24
|
+
const names = config?.backends ?? [config?.defaultBackend ?? DEFAULT_BACKEND];
|
|
25
|
+
return names.some((n) => backendFor(n)?.remote !== false);
|
|
26
|
+
}
|
|
27
|
+
|
|
9
28
|
/**
|
|
10
29
|
* `pi-dispatch sandbox` -- re-open a finished run's sandbox as an interactive shell
|
|
11
30
|
* (REQ-RESURRECTABLE-SANDBOX).
|
|
@@ -78,6 +97,23 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
|
|
|
78
97
|
return fail(err, error.message);
|
|
79
98
|
}
|
|
80
99
|
|
|
100
|
+
// #227. THE SANDBOX IS LOCAL-ONLY, and this refusal is what keeps that true rather than accidental.
|
|
101
|
+
//
|
|
102
|
+
// `buildSandboxRunArgs` is a SECOND container producer, outside the `runContainer` seam and hard-wired to
|
|
103
|
+
// this host's docker CLI. It reopens a retained job directory: `manifest.workspace` is a path on THIS
|
|
104
|
+
// machine, which for a job that ran in another venue either does not exist or exists and reproduces a
|
|
105
|
+
// run from the wrong host, silently. `INT-SANDBOX-CONTRACT` is already "a SIBLING ... never an
|
|
106
|
+
// amendment" of the container contract, and this is the clause that says which sibling.
|
|
107
|
+
//
|
|
108
|
+
// Refused BEFORE `resolveSandbox` reads anything, so a deployment that blesses a remote venue is told
|
|
109
|
+
// what is wrong rather than handed a confusing missing-directory message.
|
|
110
|
+
if (sandboxUnreachableFrom(config)) {
|
|
111
|
+
return fail(
|
|
112
|
+
err,
|
|
113
|
+
`pi-dispatch sandbox opens a shell on THIS host's docker daemon against the job's retained directory, so it cannot reach a job that ran in a remote venue (PI_BACKENDS: ${config.backends.join(", ")}). Run it on the host that ran the job.`,
|
|
114
|
+
);
|
|
115
|
+
}
|
|
116
|
+
|
|
81
117
|
const resolved = resolveSandbox({
|
|
82
118
|
jobId,
|
|
83
119
|
sandboxDir: config.sandboxDir,
|
package/src/schedules.mjs
CHANGED
|
@@ -137,7 +137,7 @@ function normalizeCronSchedule({ on, run }, path, existsSync, fleet) {
|
|
|
137
137
|
// key. A command trigger carries no flow/task at all (the validator enforces the XOR), so those two
|
|
138
138
|
// keys hold undefined here and drop at JSON serialization -- the command schedule's data is exactly
|
|
139
139
|
// kind/folder/command plus the shared fields.
|
|
140
|
-
const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
|
|
140
|
+
const data = { kind: "local", folder: run.folder, flow: run.flow, task: run.task, ...(run.command !== undefined && { command: run.command }), provider: run.provider, model: run.model, maxTurns: run.maxTurns, github: run.github, packages: run.packages, image: run.image, ...(run.backend !== undefined && { backend: run.backend }), ...(run.skillsDir !== undefined && { skillsDir: run.skillsDir }), ...(run.secrets !== undefined && { secrets: run.secrets }), ...(run.secretsProfile !== undefined && { secretsProfile: run.secretsProfile }), resume: run.resume, trigger: { id: on.id, pattern: on.pattern } };
|
|
141
141
|
// Retention only; the deterministic repeat:<id>:<millis> jobId supplies dedup, so no jobId here, and
|
|
142
142
|
// scheduler jobs are not retried (DES-CRON-VIA-BULLMQ-SCHEDULER) so no attempts/backoff.
|
|
143
143
|
const opts = { removeOnComplete: { age: 24 * 3600 }, removeOnFail: { age: 7 * 24 * 3600 } };
|