@edgehero/pi-dispatch 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +22 -1
- package/deploy/worker-env-wrapper.cmd +8 -0
- package/deploy/worker-env-wrapper.sh +58 -5
- package/package.json +1 -1
- package/src/config.mjs +33 -0
- package/src/doctor.mjs +105 -4
- package/src/env-allowlist.mjs +31 -1
- package/src/forges.mjs +14 -0
- package/src/index.mjs +11 -0
- package/src/outbox.mjs +8 -0
- package/src/processor.mjs +138 -6
- package/src/queue.mjs +14 -2
- package/src/reserved-env.mjs +40 -0
- package/src/run-container.mjs +16 -7
- package/src/run-history.mjs +40 -2
- package/src/runtime-settings.mjs +40 -0
- package/src/schedules.mjs +1 -1
- package/src/secret-profiles.mjs +119 -0
- package/src/secrets.mjs +319 -0
- package/src/session-store.mjs +294 -15
- package/src/start.mjs +19 -1
- package/src/triggers.mjs +166 -6
package/.env.example
CHANGED
|
@@ -56,10 +56,20 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
|
|
|
56
56
|
# Enforced at OPEN as well as at boot: a stale transcript is a live input to a future job, not debris
|
|
57
57
|
# PI_SESSION_MAX_BYTES= # default 8388608 (8 MiB); a transcript larger than this is not resumed; 0 = no cap
|
|
58
58
|
# Not disk hygiene -- an oversized transcript is a prefill nobody sized PI_MAX_TOKENS for
|
|
59
|
+
# PI_SESSION_MAX_AGE_DAYS= # unset/0 = no bound. How old the CONVERSATION may be, read from the session header's own timestamp
|
|
60
|
+
# A DIFFERENT CLOCK from PI_SESSIONS_TTL_DAYS, not a finer setting of it: that one reads mtime, which every COMPLETED run refreshes,
|
|
61
|
+
# so a lineage that keeps finishing work never expires however old its first turn is. This one measures from the first turn
|
|
62
|
+
# A header with no readable timestamp is refused rather than assumed young (reason: conversation-too-old)
|
|
63
|
+
# PI_SESSION_MAX_RESUME_CHAIN= # unset/0 = no bound. How many times in a row one key may be resumed before the next job starts fresh
|
|
64
|
+
# The bound a long lineage actually needs: age and size grow slowly, a chain grows once per run
|
|
65
|
+
# The count is kept whether or not the bound is set, so setting it later takes effect on the next job rather than N jobs later
|
|
66
|
+
# PI_SESSION_MAX_CONTEXT_PCT= # unset = no bound; 1-100. Refuse a resume when the saved session's context is already this full, e.g. 80
|
|
67
|
+
# A SAFETY bound before an economic one: past pi's compaction threshold a resumed job replays a model-written summary of the transcript,
|
|
68
|
+
# written while that model was reading attacker-authored text (specs/open-questions.md, OQ-003). This ceiling is the host's own, and pi's threshold stays pi's
|
|
69
|
+
# The measurement comes from the job image's runner, so it is inert until you are running an image that reports it and each key has completed one run since
|
|
59
70
|
# PI_SESSIONS_ALLOW_GH_SOURCE= # unset = a run.resume job REFUSES to mint under GITHUB_AUTH_SOURCE=gh, pre-spend
|
|
60
71
|
# That source is your whole gh login: full-scope and non-expiring, and a transcript is a FILE -- any command that echoed an auth header persists it
|
|
61
72
|
# Prefer GITHUB_AUTH_SOURCE=app or a short-expiry fine-grained PAT. Set exactly 1 to accept the trade explicitly (SECURITY.md, docs/sessions.md)
|
|
62
|
-
# Not disk hygiene -- an oversized transcript is a prefill nobody sized PI_MAX_TOKENS for
|
|
63
73
|
# PI_TRIGGERS_FILE= # ABSOLUTE path to the unified triggers.json, read by BOTH worker and receiver (a relative path resolves against the service's WorkingDirectory).
|
|
64
74
|
# Unset = cron disabled for the worker; the receiver falls back to ./triggers.json in the folder it starts from (what `pi-dispatch init` scaffolds)
|
|
65
75
|
# and refuses to start when neither exists (it holds the label/comment/pull_request trigger config)
|
|
@@ -79,6 +89,17 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
|
|
|
79
89
|
# PI_FORWARD_ENV= # comma-separated extra env var NAMES to forward into the container (e.g. a CUSTOM provider's key). Explicit allowlist, not a pass-through.
|
|
80
90
|
# GITHUB_TOKEN/GH_TOKEN are refused here -- the worker mints per-job tokens
|
|
81
91
|
|
|
92
|
+
# --- Per-trigger vault secrets (docs/secrets.md, issue #225) ---
|
|
93
|
+
# A trigger may name references ("secrets": { "STRIPE_KEY": "op://ci/stripe/api-key" }) and the profile that
|
|
94
|
+
# resolves them. The worker runs YOUR script once per reference, on the HOST, before the container starts.
|
|
95
|
+
# The job receives values; it never holds your manager's credential and cannot enumerate your vault.
|
|
96
|
+
# PI_SECRET_PROFILES= # name:/absolute/path pairs, comma separated (each entry splits on its FIRST colon, so a Windows C:\ path parses).
|
|
97
|
+
# A resolver is one line: `exec op read --no-newline "$1"`, `exec pass show "$1"`, `exec vault kv get -field=... "$1"`.
|
|
98
|
+
# Exit 2 if the reference is wrong, exit 1 if you could not reach your manager (that one retries). Unset = feature off.
|
|
99
|
+
# PI_SECRET_RESOLVER_ROOTS= # OS-path-delimited (; on Windows, : elsewhere) directories a PANEL-declared resolver may live in.
|
|
100
|
+
# Default empty = fail-closed: `/dispatch secrets add` can declare nothing, and only PI_SECRET_PROFILES above is honoured.
|
|
101
|
+
# PI_SECRET_RESOLVE_TIMEOUT_MS= # default 10000, per reference. Sits before a paid container and is multiplied by the reference count, so tighter than doctor's 30s.
|
|
102
|
+
|
|
82
103
|
PI_SCHEDULER_STALL_MAX=2 # tear down a scheduler after N consecutive stalls (money backstop)
|
|
83
104
|
|
|
84
105
|
# --- Egress policy: what a job container may reach on the network (docs/egress.md) ---
|
|
@@ -72,6 +72,14 @@ if defined ENV_SETUP (
|
|
|
72
72
|
)
|
|
73
73
|
)
|
|
74
74
|
|
|
75
|
+
REM WEAKER THAN THE .sh TWIN ON SIGNALS, deliberately and stated rather than discovered (issue #221).
|
|
76
|
+
REM cmd has no `trap`, so there is no wrapper-level handling of a stop that arrives while `.env` is being
|
|
77
|
+
REM read or while the setup script above is still running: whatever the service manager does to the tree
|
|
78
|
+
REM is what happens. nssm stops with a console event to the process group (AppStopMethodConsole), so the
|
|
79
|
+
REM worker is reached directly rather than through this file, which is why the .sh twin's forwarding
|
|
80
|
+
REM problem has no equivalent here. The asymmetry is recorded in DES-SERVICE-ENV-SETUP-SEAM and is not
|
|
81
|
+
REM closed.
|
|
82
|
+
REM
|
|
75
83
|
REM The argv runs verbatim -- absolute node, absolute script, composed by `pi-dispatch service` (see
|
|
76
84
|
REM the .sh twin for the whole contract).
|
|
77
85
|
%*
|
|
@@ -36,6 +36,34 @@ if [ "$#" -eq 0 ]; then
|
|
|
36
36
|
exit 1
|
|
37
37
|
fi
|
|
38
38
|
|
|
39
|
+
# STOP HANDLING IS ARMED HERE, above everything below that can block (issue #221). It closes two windows,
|
|
40
|
+
# both of which used to swallow a stop in silence.
|
|
41
|
+
#
|
|
42
|
+
# Until this line TERM/INT carry their DEFAULT disposition, and the sourcing below can take arbitrarily
|
|
43
|
+
# long: PI_ENV_SETUP is an operator's secrets manager, so docs/secrets.md's own worked example makes a
|
|
44
|
+
# network round trip inside it. A stop landing there killed this shell where it stood, mid-preparation,
|
|
45
|
+
# with nothing anywhere saying the environment had been half-built. That is reachable from this project's
|
|
46
|
+
# own CLI, not just from the daemon: `pi-dispatch service stop` on macOS is `launchctl kill SIGTERM` at
|
|
47
|
+
# this pid.
|
|
48
|
+
#
|
|
49
|
+
# The other window is two instructions wide, and is closed by the re-send after `child=$!` below. The
|
|
50
|
+
# handler is a FUNCTION rather than a trap string because it is installed twice -- here, and again after
|
|
51
|
+
# the sourcing -- and one behaviour spelled out in two places is one behaviour that can drift.
|
|
52
|
+
signaled=0
|
|
53
|
+
child=
|
|
54
|
+
wrapper_on_stop() {
|
|
55
|
+
signaled=1
|
|
56
|
+
# `child` is empty until the fork below has been assigned, and `kill -TERM ""` kills nothing and
|
|
57
|
+
# fails silently, so a stop arriving before then has no pid to reach. It is not lost: the re-send
|
|
58
|
+
# after `child=$!` re-delivers it, and the launch gate refuses to start at all if nothing was
|
|
59
|
+
# started yet.
|
|
60
|
+
[ -n "$child" ] && kill -TERM "$child" 2>/dev/null
|
|
61
|
+
# Never leave a nonzero status behind. `rc=$?` is read immediately after the `wait` this interrupts,
|
|
62
|
+
# and the double wait at the bottom keys on rc >= 128.
|
|
63
|
+
return 0
|
|
64
|
+
}
|
|
65
|
+
trap wrapper_on_stop TERM INT
|
|
66
|
+
|
|
39
67
|
# The env-setup seam (issue #209): `pi-dispatch service render|install --env-setup <path>` puts an
|
|
40
68
|
# operator-typed path here -- the plist's EnvironmentVariables dict on macOS, nssm's AppEnvironmentExtra
|
|
41
69
|
# on Windows -- so a secrets manager can fill this process's environment without anyone hand-editing a
|
|
@@ -74,6 +102,25 @@ if [ -n "$env_setup" ]; then
|
|
|
74
102
|
set +a
|
|
75
103
|
fi
|
|
76
104
|
|
|
105
|
+
# RE-ASSERTED after the sourcing, and this is not belt-and-braces. A sourced script runs in THIS shell,
|
|
106
|
+
# so a `trap ... TERM` inside one REPLACES the handler above and the drain silently disappears -- a
|
|
107
|
+
# manager's cleanup helper does exactly that. One line restores it. What it cannot undo is a script that
|
|
108
|
+
# IGNORES TERM (`trap '' TERM`): a signal discarded while it was ignored is already gone, and the child
|
|
109
|
+
# forked below would inherit SIG_IGN and be unable to trap TERM at all. That is why docs/secrets.md now
|
|
110
|
+
# tells operators not to touch signals in a setup script.
|
|
111
|
+
trap wrapper_on_stop TERM INT
|
|
112
|
+
|
|
113
|
+
# A stop that arrived while the environment was being prepared is honoured by NOT STARTING. Launching now
|
|
114
|
+
# would hand the service manager a worker it has already asked to go away: it would reserve a budget slot
|
|
115
|
+
# and take a job, and then need a drain nobody is waiting for. Exit 0 because 0 is the only code launchd's
|
|
116
|
+
# KeepAlive/SuccessfulExit=false leaves stopped -- the same reason the exit-2 conversion at the bottom
|
|
117
|
+
# exists. Not 2, because nothing was refused; not 1, because nothing failed; the manager's own instruction
|
|
118
|
+
# was carried out, and this says so rather than exiting mute.
|
|
119
|
+
if [ "$signaled" -eq 1 ]; then
|
|
120
|
+
echo "worker-env-wrapper: stopped before the worker started -- a stop signal arrived while the environment was being prepared, so the command was never launched; exiting 0 (nothing to restart)" >&2
|
|
121
|
+
exit 0
|
|
122
|
+
fi
|
|
123
|
+
|
|
77
124
|
# `exec` is deliberately GONE here (it used to hand this shell's pid straight to node): intercepting
|
|
78
125
|
# the exit code needs a parent still alive after node exits. launchd's KeepAlive/SuccessfulExit=false
|
|
79
126
|
# relaunches ANY nonzero exit -- including EXIT_POLICY (2, worker/src/exit-code.mjs), the determinate
|
|
@@ -81,13 +128,19 @@ fi
|
|
|
81
128
|
# deliberately never retry. A relaunch loop against a paid provider is a bill, so the conversion at
|
|
82
129
|
# the bottom turns exit 2 into the clean exit KeepAlive leaves stopped.
|
|
83
130
|
#
|
|
84
|
-
# SIGTERM still reaches node without exec: the
|
|
85
|
-
# a foreground command in sh, which blocks trap delivery) is interruptible by a trapped
|
|
86
|
-
# forwarding is immediate and node gets its full graceful drain.
|
|
87
|
-
signaled=0
|
|
88
|
-
trap 'signaled=1; kill -TERM "$child" 2>/dev/null' TERM INT
|
|
131
|
+
# SIGTERM still reaches node without exec: the handler armed at the top forwards TERM/INT to the child,
|
|
132
|
+
# and `wait` (unlike a foreground command in sh, which blocks trap delivery) is interruptible by a trapped
|
|
133
|
+
# signal, so the forwarding is immediate and node gets its full graceful drain.
|
|
89
134
|
"$@" &
|
|
90
135
|
child=$!
|
|
136
|
+
# THE FORK WINDOW (issue #221). `$!` is only readable in the parent AFTER the fork, so between the two
|
|
137
|
+
# lines above a child exists and its pid does not. A stop landing there ran the handler with nothing to
|
|
138
|
+
# forward to, set `signaled`, and was then never looked at again -- so this wrapper waited out the
|
|
139
|
+
# command's ENTIRE natural lifetime while the service manager believed it had asked it to stop. Re-sending
|
|
140
|
+
# once the pid is known costs one `[` on the healthy path and is the whole difference between a graceful
|
|
141
|
+
# drain and a hang as long as the job. Issue #207 found this same drop through the test that saw it and
|
|
142
|
+
# fixed only the test; #221 is the same window firing through a different one.
|
|
143
|
+
[ "$signaled" -eq 1 ] && kill -TERM "$child" 2>/dev/null
|
|
91
144
|
wait "$child"
|
|
92
145
|
rc=$?
|
|
93
146
|
# The double wait is load-bearing: a trapped signal interrupts the FIRST wait early (rc = 128+signum)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
package/src/config.mjs
CHANGED
|
@@ -9,6 +9,7 @@ import { existsSync } from "node:fs";
|
|
|
9
9
|
import { delimiter } from "node:path";
|
|
10
10
|
import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
|
|
11
11
|
import { MINTED_TOKEN_VARS } from "./forges.mjs";
|
|
12
|
+
import { parseSecretProfiles } from "./secret-profiles.mjs";
|
|
12
13
|
|
|
13
14
|
export function configError(message) {
|
|
14
15
|
const error = new Error(message);
|
|
@@ -258,10 +259,42 @@ export function loadConfig(env = process.env, { fileExists = existsSync } = {})
|
|
|
258
259
|
// A bound on how large a transcript may be before it stops being resumed. Not disk hygiene: an
|
|
259
260
|
// oversized transcript is a prefill an operator never sized PI_MAX_TOKENS for.
|
|
260
261
|
sessionMaxBytes: nonNegativeInt(env, "PI_SESSION_MAX_BYTES", 8 * 1024 * 1024), // 0 = no cap
|
|
262
|
+
// How old the CONVERSATION may be, which is a different clock from sessionsTtlDays above and not a
|
|
263
|
+
// finer setting of it: the TTL reads the transcript's mtime, which the PROMOTE rename refreshes (the
|
|
264
|
+
// resolve copy does not, measured -- copyFileSync stamps the destination), so it measures time since
|
|
265
|
+
// the last COMPLETED run on this key. A key whose runs keep completing never expires however old its
|
|
266
|
+
// first turn is. This one reads the session header's own timestamp.
|
|
267
|
+
// OFF by default (0) rather than defaulted to a number: an age an operator did not choose is an
|
|
268
|
+
// opinion about their lineages that this project has no basis for.
|
|
269
|
+
sessionMaxAgeDays: nonNegativeInt(env, "PI_SESSION_MAX_AGE_DAYS", 0), // 0 = no age bound
|
|
270
|
+
// How many times in a row one key may be resumed before the next job starts fresh. The bound a long
|
|
271
|
+
// lineage actually needs: age and size both grow slowly while a chain grows once per run.
|
|
272
|
+
sessionMaxResumeChain: nonNegativeInt(env, "PI_SESSION_MAX_RESUME_CHAIN", 0), // 0 = no chain bound
|
|
273
|
+
// How full the saved context may be before a resume is refused, as a percentage of the model's own
|
|
274
|
+
// window. A PERCENTAGE, so `optionalBoundedInt` on softHoldPct's precedent rather than the 0 = off
|
|
275
|
+
// sentinel its two neighbours use: 0% would mean "never resume anything", which is a different
|
|
276
|
+
// request from "no bound", and 101 is a typo rather than a ceiling.
|
|
277
|
+
sessionMaxContextPct: optionalBoundedInt(env, "PI_SESSION_MAX_CONTEXT_PCT", 1, 100), // null = no context bound
|
|
261
278
|
chainDepthMax: nonNegativeInt(env, "PI_CHAIN_DEPTH_MAX", CHAIN_DEPTH_MAX_DEFAULT), // DES-JOB-OUTBOX-CHAINING; 0 = chaining kill-switch (fail-closed)
|
|
262
279
|
chainMaxPerJob: nonNegativeInt(env, "PI_CHAIN_MAX_PER_JOB", CHAIN_MAX_PER_JOB_DEFAULT), // INT-OUTBOX-CONTRACT: max request-<n>.json collected per parent
|
|
263
280
|
dispatchRunPerHour: nonNegativeInt(env, "PI_DISPATCH_RUN_PER_HOUR", 3), // DES-ADMIN-VIA-PI-EXTENSION; 0 = disable dispatch_run
|
|
264
281
|
dispatchRunRoots: delimitedList(env.PI_DISPATCH_RUN_ROOTS), // DES-AI-TRIGGER-FLOW-GATE: default [] fails closed — no folder passes, dispatch_run refuses everything
|
|
282
|
+
// REQ-TRIGGER-SECRETS. The operator's declared resolvers, `name:absolute-path` pairs, comma separated.
|
|
283
|
+
// Each entry splits on its FIRST colon so a Windows `C:\...` path survives -- the same drive-letter
|
|
284
|
+
// hazard `delimitedList` above exists for, arriving from the other side. Unset = the feature is off and
|
|
285
|
+
// any trigger naming secrets refuses pre-spend, which is why there is no default profile to fall into.
|
|
286
|
+
secretProfiles: parseSecretProfiles(env.PI_SECRET_PROFILES),
|
|
287
|
+
// The directories a resolver may live in. Default [] FAILS CLOSED exactly as dispatchRunRoots does, and
|
|
288
|
+
// for a sharper version of its reason: this bounds paths that can arrive from the settings overlay,
|
|
289
|
+
// which is not the reviewed artifact `triggers.json` is. Unset means the panel can declare no profile
|
|
290
|
+
// at all and only PI_SECRET_PROFILES above is honoured. Env-only, never the overlay and never the
|
|
291
|
+
// deployment pointer: `deployment-pointer.mjs` already refuses to carry PI_DISPATCH_RUN_ROOTS
|
|
292
|
+
// "because a pointer that could widen the AI-run folder allowlist would be a second, unreviewed door",
|
|
293
|
+
// and a bound that can be widened from the surface it bounds is not a bound.
|
|
294
|
+
secretResolverRoots: delimitedList(env.PI_SECRET_RESOLVER_ROOTS),
|
|
295
|
+
// Per-reference ceiling. Tighter than doctor's 30s on purpose: this runs before a paid container, is
|
|
296
|
+
// multiplied by the reference count, and holds a PI_CONCURRENCY slot while it waits.
|
|
297
|
+
secretResolveTimeoutMs: positiveInt(env, "PI_SECRET_RESOLVE_TIMEOUT_MS", 10000),
|
|
265
298
|
github: { ...loadGitHubAuth(env, fileExists), allowGhResume: env.PI_SESSIONS_ALLOW_GH_SOURCE === "1" },
|
|
266
299
|
gitlab: loadGitLabAuth(env),
|
|
267
300
|
forgejo: loadForgejoAuth(env),
|
package/src/doctor.mjs
CHANGED
|
@@ -46,7 +46,7 @@
|
|
|
46
46
|
*/
|
|
47
47
|
import { chmodSync, closeSync, existsSync, lstatSync, mkdirSync, mkdtempSync, openSync, readdirSync, readFileSync, readSync, rmSync, statSync } from "node:fs";
|
|
48
48
|
import { homedir, tmpdir } from "node:os";
|
|
49
|
-
import { dirname, join } from "node:path";
|
|
49
|
+
import { dirname, join, delimiter } from "node:path";
|
|
50
50
|
import { fileURLToPath } from "node:url";
|
|
51
51
|
import { spawn as nodeSpawn } from "node:child_process";
|
|
52
52
|
import { defaultSandboxDir, globalExtensionsEnabled } from "./config.mjs";
|
|
@@ -58,6 +58,7 @@ import { copySkillTree } from "./copy-tree.mjs";
|
|
|
58
58
|
import { SKILL_NAME_RE } from "./flow-gate.mjs";
|
|
59
59
|
import { DEFAULT_EGRESS_PROXY, egressArmed } from "./egress.mjs";
|
|
60
60
|
import { installedUnitPaths, readUnitSeam } from "./service.mjs";
|
|
61
|
+
import { parseSecretProfiles } from "./secret-profiles.mjs";
|
|
61
62
|
import { parseTriggers } from "./triggers.mjs";
|
|
62
63
|
|
|
63
64
|
const NODE_FLOOR = [22, 19]; // pi's engine floor (22.19.0)
|
|
@@ -263,7 +264,7 @@ export async function collectChecks(env, seams) {
|
|
|
263
264
|
// image checks just below, and `optingOut`/`requiring` colour the staged-packages lines further down.
|
|
264
265
|
// `optingOut` counts the only value that withholds the staged set; `requiring` counts an explicit
|
|
265
266
|
// run.packages: true, which arms nothing any more but is still an operator statement of intent.
|
|
266
|
-
const { requiring, optingOut, resuming, replicating, instructing, commands, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
|
|
267
|
+
const { requiring, optingOut, resuming, replicating, instructing, commands, secreting, secretProfiles, localSecretFolders, images, skillsDirs, forges, repositories, flows, parseError, path: triggersFilePath } = readTriggerFacts(env, fileExists, cwd);
|
|
267
268
|
// FIRST, and fail rather than warn: every check below this line reads counts that a parse failure
|
|
268
269
|
// zeroed, so a green run here would be reporting on a file nobody could read. The receiver loads this
|
|
269
270
|
// file unconditionally and refuses to start without it, which is the consequence worth naming.
|
|
@@ -1096,6 +1097,52 @@ export async function collectChecks(env, seams) {
|
|
|
1096
1097
|
});
|
|
1097
1098
|
}
|
|
1098
1099
|
|
|
1100
|
+
// REQ-TRIGGER-SECRETS. Only reported when a trigger actually binds one, on the run.resume block's
|
|
1101
|
+
// reasoning below: a deployment that uses no secrets should not be told about a variable it has no
|
|
1102
|
+
// reason to set.
|
|
1103
|
+
if (secreting > 0) {
|
|
1104
|
+
const declared = parseSecretProfilesSafe(env.PI_SECRET_PROFILES);
|
|
1105
|
+
const names = Object.keys(declared).sort();
|
|
1106
|
+
if (declared.error) {
|
|
1107
|
+
checks.push({ ok: false, label: "PI_SECRET_PROFILES does not parse", fix: `${declared.error} -- the worker refuses to boot until this is fixed, rather than dropping the entry and leaving you a profile you believe is wired` });
|
|
1108
|
+
} else {
|
|
1109
|
+
// A HARD FAIL, not a warning, and worded like the run.resume/PI_SESSIONS_DIR check below for the
|
|
1110
|
+
// same reason: these jobs refuse pre-spend until it is set, deliberately, rather than running
|
|
1111
|
+
// without their secrets and looking like they worked.
|
|
1112
|
+
const missing = secretProfiles.filter((name) => !(name in declared));
|
|
1113
|
+
checks.push({
|
|
1114
|
+
ok: missing.length === 0,
|
|
1115
|
+
label: missing.length === 0 ? `${secreting} trigger(s) bind secrets, and every profile they name is declared` : `${secreting} trigger(s) bind secrets, but ${missing.length} named profile(s) are not declared: ${missing.join(", ")}`,
|
|
1116
|
+
fix: `declare them in PI_SECRET_PROFILES as name:/absolute/path pairs (a resolver is one line, e.g. \`exec op read --no-newline "$1"\`) -- these jobs refuse pre-spend until you do`,
|
|
1117
|
+
});
|
|
1118
|
+
// The declared table, so an operator sees what is wired without reading .env. NAMES and paths only:
|
|
1119
|
+
// this is doctor's own stdout on the operator's host, not a public issue comment.
|
|
1120
|
+
if (names.length > 0) {
|
|
1121
|
+
checks.push({ ok: true, label: `Secret resolver profiles declared: ${names.map((n) => `${n} -> ${declared[n]}`).join(", ")}` });
|
|
1122
|
+
}
|
|
1123
|
+
// The panel-authoring bound. Unset is the SAFE default rather than a defect, so this is a fact line
|
|
1124
|
+
// when closed and a disclosure when open.
|
|
1125
|
+
const roots = (env.PI_SECRET_RESOLVER_ROOTS ?? "").split(delimiter).map((r) => r.trim()).filter(Boolean);
|
|
1126
|
+
checks.push({
|
|
1127
|
+
ok: true,
|
|
1128
|
+
...(roots.length === 0
|
|
1129
|
+
? { label: "PI_SECRET_RESOLVER_ROOTS is unset, so only PI_SECRET_PROFILES declares resolvers (the panel can declare none)" }
|
|
1130
|
+
: { warn: true, label: `PI_SECRET_RESOLVER_ROOTS admits panel-declared resolvers under: ${roots.join(", ")}`, fix: "keep those directories writable by nobody but the account the worker runs as: whoever can write a resolver there can run code as the worker" }),
|
|
1131
|
+
});
|
|
1132
|
+
}
|
|
1133
|
+
// The local-workspace disclosure. Not a failure: a nightly deploy binding a secret is exactly what
|
|
1134
|
+
// this feature is for. But a local job's /workspace IS the folder, read-write and un-cloned, so an
|
|
1135
|
+
// agent that persists a credential to make its next command simpler writes it into a real repository.
|
|
1136
|
+
if (localSecretFolders.length > 0) {
|
|
1137
|
+
checks.push({
|
|
1138
|
+
ok: true,
|
|
1139
|
+
warn: true,
|
|
1140
|
+
label: `${localSecretFolders.length} local trigger(s) bind secrets and run IN the operator's own folder: ${localSecretFolders.join(", ")}`,
|
|
1141
|
+
fix: "a local job edits that folder in place, so a credential the agent writes to .env, .netrc or .git-credentials lands in your real repository (and survives in a retained sandbox for PI_SANDBOX_RETENTION_HOURS). Nothing scans for that: keep those folders out of anything you push",
|
|
1142
|
+
});
|
|
1143
|
+
}
|
|
1144
|
+
}
|
|
1145
|
+
|
|
1099
1146
|
// REQ-RESUMABLE-SESSION. Only reported when a trigger actually asked for it: a deployment that does
|
|
1100
1147
|
// not use resume should not be told about a directory it has no reason to create.
|
|
1101
1148
|
if (resuming > 0) {
|
|
@@ -1132,8 +1179,37 @@ export async function collectChecks(env, seams) {
|
|
|
1132
1179
|
ok: true,
|
|
1133
1180
|
warn: true,
|
|
1134
1181
|
label: `${resuming} trigger(s) persist agent transcripts to ${sessionsDir} -- PII-bearing, host-only, never committed`,
|
|
1135
|
-
fix: "confirm it is outside every git repo and on a disk you would put issue text on; PI_SESSIONS_TTL_DAYS
|
|
1182
|
+
fix: "confirm it is outside every git repo and on a disk you would put issue text on; PI_SESSIONS_TTL_DAYS, PI_SESSION_MAX_AGE_DAYS, PI_SESSION_MAX_RESUME_CHAIN and PI_SESSION_MAX_CONTEXT_PCT each bound a different thing about how much history one key accumulates (docs/sessions.md)",
|
|
1183
|
+
});
|
|
1184
|
+
// Which of the four bounds are actually on, as a FACT LINE rather than a warning: how long a
|
|
1185
|
+
// lineage may run is an operator's call, not a defect, and doctor's warnings are for things that
|
|
1186
|
+
// need a decision. The line exists because these knobs are unset by default and silent when
|
|
1187
|
+
// unset, so the only way to tell a deliberate "no bound" from a forgotten one is to print it.
|
|
1188
|
+
const bounds = [
|
|
1189
|
+
["PI_SESSIONS_TTL_DAYS", env.PI_SESSIONS_TTL_DAYS, "14"],
|
|
1190
|
+
["PI_SESSION_MAX_AGE_DAYS", env.PI_SESSION_MAX_AGE_DAYS, "off"],
|
|
1191
|
+
["PI_SESSION_MAX_RESUME_CHAIN", env.PI_SESSION_MAX_RESUME_CHAIN, "off"],
|
|
1192
|
+
["PI_SESSION_MAX_CONTEXT_PCT", env.PI_SESSION_MAX_CONTEXT_PCT, "off"],
|
|
1193
|
+
];
|
|
1194
|
+
checks.push({
|
|
1195
|
+
ok: true,
|
|
1196
|
+
label: `Resume bounds: ${bounds.map(([name, value, fallback]) => `${name}=${value === undefined || value === "" ? fallback : value}`).join(", ")}`,
|
|
1136
1197
|
});
|
|
1198
|
+
// The one bound that can be set and still do nothing, and the operator cannot see it from here.
|
|
1199
|
+
// Its measurement is reported by the JOB IMAGE's runner (INT-RUNNER-EXIT-CODE-PROTOCOL), so an
|
|
1200
|
+
// image older than that field reports none, the gate passes on no measurement by design, and the
|
|
1201
|
+
// bound is inert with nothing anywhere saying so. There is deliberately no image capability to
|
|
1202
|
+
// check against -- capabilities are an inclusion list for what the host DEMANDS of an image, and
|
|
1203
|
+
// telemetry is not that -- so this warning is the whole detection surface, which is exactly why
|
|
1204
|
+
// it exists rather than being left to a doc.
|
|
1205
|
+
if (env.PI_SESSION_MAX_CONTEXT_PCT) {
|
|
1206
|
+
checks.push({
|
|
1207
|
+
ok: true,
|
|
1208
|
+
warn: true,
|
|
1209
|
+
label: `PI_SESSION_MAX_CONTEXT_PCT=${env.PI_SESSION_MAX_CONTEXT_PCT} needs a job image whose runner reports context usage`,
|
|
1210
|
+
fix: `an older image reports none, and a bound with no measurement passes rather than guessing, so on such an image this bound does nothing at all. On an image that does report one the reading is kept whether or not the bound is set, so it applies from the next job. Each run's own record (${env.PI_LOGS_DIR || "the logs directory"}/<jobId>.json) carries session.reason, which names the gate that refused`,
|
|
1211
|
+
});
|
|
1212
|
+
}
|
|
1137
1213
|
}
|
|
1138
1214
|
}
|
|
1139
1215
|
|
|
@@ -1381,8 +1457,22 @@ async function repoFlowAtHead(spawn, folder, flow) {
|
|
|
1381
1457
|
return mode === "100644" && type === "blob" ? "present" : "absent";
|
|
1382
1458
|
}
|
|
1383
1459
|
|
|
1460
|
+
/**
|
|
1461
|
+
* `parseSecretProfiles`, but doctor never throws. A malformed PI_SECRET_PROFILES is a finding to REPORT,
|
|
1462
|
+
* not a reason for the diagnostic tool to die: the operator running doctor is very likely running it
|
|
1463
|
+
* BECAUSE the worker refused to boot on that exact line, and a stack trace instead of a check is the least
|
|
1464
|
+
* useful possible answer. Returns the table, or `{ error }` carrying the parser's own message.
|
|
1465
|
+
*/
|
|
1466
|
+
function parseSecretProfilesSafe(raw) {
|
|
1467
|
+
try {
|
|
1468
|
+
return parseSecretProfiles(raw);
|
|
1469
|
+
} catch (err) {
|
|
1470
|
+
return { error: err?.message ?? "unparseable" };
|
|
1471
|
+
}
|
|
1472
|
+
}
|
|
1473
|
+
|
|
1384
1474
|
function readTriggerFacts(env, fileExists, cwd) {
|
|
1385
|
-
const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
|
|
1475
|
+
const none = { requiring: 0, optingOut: 0, resuming: 0, replicating: 0, instructing: 0, commands: 0, secreting: 0, secretProfiles: [], localSecretFolders: [], images: [], skillsDirs: [], forges: [], repositories: [], flows: [], parseError: null, path: null };
|
|
1386
1476
|
try {
|
|
1387
1477
|
// Unset falls back to ./triggers.json in cwd, MIRRORING the receiver's own default
|
|
1388
1478
|
// (receiver/src/config.mjs) -- the two must read the same file, or doctor preflights a deployment
|
|
@@ -1403,6 +1493,17 @@ function readTriggerFacts(env, fileExists, cwd) {
|
|
|
1403
1493
|
// tuple list already filters to `typeof f.flow === "string"`, so a command trigger drops out
|
|
1404
1494
|
// of the flow-tier probes naturally -- no exclusion needed there.
|
|
1405
1495
|
commands: triggers.filter((t) => typeof t.run.command === "string").length,
|
|
1496
|
+
// REQ-TRIGGER-SECRETS. Counted beside `instructing` for its reason: a per-trigger choice that
|
|
1497
|
+
// changes what every job of it can reach, and one that lives only in triggers.json.
|
|
1498
|
+
secreting: triggers.filter((t) => t.run.secrets !== undefined).length,
|
|
1499
|
+
// The distinct profile NAMES the file selects, deduped like `images`/`skillsDirs`: the checks below
|
|
1500
|
+
// cost a stat each, and two triggers naming one profile are one question. `default` is substituted
|
|
1501
|
+
// for an absent field so the table answers what the worker will actually look up.
|
|
1502
|
+
secretProfiles: [...new Set(triggers.filter((t) => t.run.secrets !== undefined).map((t) => t.run.secretsProfile ?? "default"))].sort(),
|
|
1503
|
+
// LOCAL triggers that bind secrets, by folder. A local job's /workspace IS this folder, bind-mounted
|
|
1504
|
+
// read-write with no clone, so a credential an agent writes into .env lands in the operator's real
|
|
1505
|
+
// repository rather than a temp dir that gets swept. Deduped for skillsDirs' reason.
|
|
1506
|
+
localSecretFolders: [...new Set(triggers.filter((t) => t.run.secrets !== undefined && t.run.kind === "local" && typeof t.run.folder === "string").map((t) => t.run.folder))].sort(),
|
|
1406
1507
|
optingOut: triggers.filter((t) => t.run.packages === false).length,
|
|
1407
1508
|
images: [...new Set(triggers.map((t) => t.run.image).filter((i) => typeof i === "string"))].sort(),
|
|
1408
1509
|
// REQ-PER-TRIGGER-SKILLS. The distinct host directories the file names, deduped like `images`,
|
package/src/env-allowlist.mjs
CHANGED
|
@@ -111,7 +111,7 @@ function resolveEnvName(provider, cred) {
|
|
|
111
111
|
* `allowGlobalExtensions` defaults to TRUE here, matching loadConfig's default (REQ-GLOBAL-PI-OVERLAY): a
|
|
112
112
|
* caller that says nothing gets the operator's staged setup, and only an explicit `false` withholds it.
|
|
113
113
|
*/
|
|
114
|
-
export function buildContainerEnv({ provider, model, maxTurns, maxTokens, jobId, githubToken, forgeKind, forgeHosts = {}, hostEnv, allowGlobalExtensions = true, packagePaths = [], forwardEnv = [], sessionFile = null, flow = null, command = null, authFromPi = false, egress = false, egressProxy, agentDir, readFile = readFileSync }) {
|
|
114
|
+
export function buildContainerEnv({ provider, model, maxTurns, maxTokens, jobId, githubToken, forgeKind, forgeHosts = {}, hostEnv, allowGlobalExtensions = true, packagePaths = [], forwardEnv = [], secrets = {}, sessionFile = null, flow = null, command = null, authFromPi = false, egress = false, egressProxy, agentDir, readFile = readFileSync }) {
|
|
115
115
|
// The provider credential(s), by pi's expected variable name(s) -- from the worker env, or (when
|
|
116
116
|
// PI_AUTH_FROM_PI is set and the env has none) host-side from pi's auth.json. Throws (config) if
|
|
117
117
|
// neither source yields one, which the processor turns into a pre-spend refusal.
|
|
@@ -182,6 +182,36 @@ export function buildContainerEnv({ provider, model, maxTurns, maxTokens, jobId,
|
|
|
182
182
|
if (hostEnv[name] !== undefined) env[name] = hostEnv[name];
|
|
183
183
|
}
|
|
184
184
|
|
|
185
|
+
// The trigger's own secrets (REQ-TRIGGER-SECRETS), resolved HOST-SIDE by the processor before anything
|
|
186
|
+
// spent and injected exactly the way the provider credential is -- never a vault credential handed to the
|
|
187
|
+
// container to fetch them itself. `docs/secrets.md`'s rule survives intact: what crosses the boundary is a
|
|
188
|
+
// value, and the thing that can FETCH values stays on the host.
|
|
189
|
+
//
|
|
190
|
+
// AFTER the PI_FORWARD_ENV loop, for that loop's own stated reason: a name on the operator's blanket host
|
|
191
|
+
// list must not silently outrank the specific reference this trigger declared.
|
|
192
|
+
//
|
|
193
|
+
// BEFORE the egress assign, and that direction is deliberate rather than incidental. A secret named
|
|
194
|
+
// HTTPS_PROXY that WON would point this job away from the proxy its --internal network was built around,
|
|
195
|
+
// while reading exactly like the control working -- an OUTAGE dressed as a policy. config.mjs refuses
|
|
196
|
+
// those names in PI_FORWARD_ENV outright while the policy is armed, for the same reason.
|
|
197
|
+
//
|
|
198
|
+
// BEFORE the mint below too, so the per-job scoped token still wins. A vault-supplied GITHUB_TOKEN overwriting
|
|
199
|
+
// the mint would hand every container a long-lived operator credential: CONST-TOKEN-SCOPED-PER-JOB
|
|
200
|
+
// defeated by a config line, which is the inversion forwardEnvList refuses at boot.
|
|
201
|
+
//
|
|
202
|
+
// Ordering is the BACKSTOP, not the gate. parseTriggers refuses every statically knowable one of these
|
|
203
|
+
// names at load, and the processor refuses the provider's own credential variables and the PI_FORWARD_ENV
|
|
204
|
+
// names pre-spend, where the resolved provider and the host env are in hand. This is the same division of
|
|
205
|
+
// labour the minted token already keeps ("and loadConfig refuses those names at load anyway").
|
|
206
|
+
//
|
|
207
|
+
// A LOOP rather than Object.assign, so a non-string or empty value becomes an ABSENT variable rather than
|
|
208
|
+
// `NAME=`: docker-run skips `undefined` but not `""`, and "never an empty string" is the rule PI_PACKAGES,
|
|
209
|
+
// PI_SESSION_FILE and PI_FLOW already keep. The resolver guarantees non-empty; this is the defense in
|
|
210
|
+
// depth at the DI seam that the empty-token guard keeps for the mint.
|
|
211
|
+
for (const [name, value] of Object.entries(secrets ?? {})) {
|
|
212
|
+
if (typeof value === "string" && value !== "") env[name] = value;
|
|
213
|
+
}
|
|
214
|
+
|
|
185
215
|
// The shipped egress policy's variables (REQ-EGRESS-ALLOWLIST), AFTER the PI_FORWARD_ENV loop so a
|
|
186
216
|
// forwarded name can never override them -- the same ordering, for the same reason, as the minted token
|
|
187
217
|
// below (and loadConfig refuses those names outright while the policy is armed anyway).
|
package/src/forges.mjs
CHANGED
|
@@ -135,6 +135,20 @@ export function forgeSpec(kind) {
|
|
|
135
135
|
*/
|
|
136
136
|
export const MINTED_TOKEN_VARS = new Set(FORGE_KINDS.flatMap((kind) => FORGES[kind].tokenVars));
|
|
137
137
|
|
|
138
|
+
/**
|
|
139
|
+
* Every environment variable name any forge's SELF-HOSTED INSTANCE URL can land in. A sibling of
|
|
140
|
+
* `MINTED_TOKEN_VARS` and derived the same way, from the table rather than by hand, so a forge added
|
|
141
|
+
* later cannot be missed here either.
|
|
142
|
+
*
|
|
143
|
+
* Separate from `MINTED_TOKEN_VARS` because `hostVar` is a separate column: the token set is
|
|
144
|
+
* `tokenVars` only, so a name like `GITLAB_HOST` is in NEITHER set until this one exists. That gap is
|
|
145
|
+
* not theoretical -- `buildContainerEnv` writes this variable after the mint, so a trigger field able
|
|
146
|
+
* to name it would be silently overwritten, and the job would run without the value it asked for on a
|
|
147
|
+
* clean exit 0. github's row is `null` and contributes nothing, which is why the filter is here and
|
|
148
|
+
* not at the call site.
|
|
149
|
+
*/
|
|
150
|
+
export const FORGE_HOST_VARS = new Set(FORGE_KINDS.map((kind) => FORGES[kind].hostVar).filter((name) => typeof name === "string"));
|
|
151
|
+
|
|
138
152
|
/**
|
|
139
153
|
* The separator between a repo label and a target number, for this forge and this target type: the
|
|
140
154
|
* forge's own notation for a pull/merge request, and `#` for an issue everywhere.
|
package/src/index.mjs
CHANGED
|
@@ -101,6 +101,17 @@ export function makeProcessor({ cancelJob, stopContainer, redis, getSettings, ap
|
|
|
101
101
|
tokenCap: settings.dailyTokenCap,
|
|
102
102
|
...deps,
|
|
103
103
|
runContainer: (ctx) => deps.runContainer({ ...ctx, name, signal }),
|
|
104
|
+
// REQ-TRIGGER-SECRETS. The resolver runs INSIDE the 30-minute kill timer armed above, so it has
|
|
105
|
+
// to be abortable for the same reason runContainer does: a resolver blocking on an unreachable
|
|
106
|
+
// vault would otherwise hold its slot until its own timeout, and an abort landing mid-resolution
|
|
107
|
+
// would neither stop it nor keep the job from going on to mint, clone and reserve budget for a
|
|
108
|
+
// container that runContainer will refuse at entry anyway. Injected here, mirroring runContainer,
|
|
109
|
+
// because `signal` exists only in this scope. Omitted when unwired so a bare processor keeps
|
|
110
|
+
// runJob's own fail-closed default.
|
|
111
|
+
// `secretProfiles` is the OVERLAY half of the resolver table, read this job-start with the ten
|
|
112
|
+
// tunables above. It is bound here rather than at construction for the reason the overlay exists:
|
|
113
|
+
// an operator who declares a profile in the panel must not have to restart the worker.
|
|
114
|
+
...(deps.resolveSecrets ? { resolveSecrets: (j) => deps.resolveSecrets(j, { signal, overlayProfiles: settings.secretProfiles ?? {} }) } : {}),
|
|
104
115
|
// collectChain (INT-OUTBOX-CONTRACT) reads the completed parent's REAL BullMQ job: its `.id`
|
|
105
116
|
// (the parent id children carry) and `.data` (kind/chainDepth). runJob's own `job` is the
|
|
106
117
|
// effectiveJob -- a spread of job.data with no `.id`/`.data` -- so inject the real wrapper here,
|
package/src/outbox.mjs
CHANGED
|
@@ -176,6 +176,14 @@ export function makeCollectChain({ queue, enqueue = enqueueLocalJob, readFlowGat
|
|
|
176
176
|
// same operator's flows and, without them, would look up a skill that is not there, write a
|
|
177
177
|
// plausible report and exit 0. It is NOT part of chainedJobId, for the reason stated below.
|
|
178
178
|
skillsDir: job.data?.skillsDir,
|
|
179
|
+
// `secrets`/`secretsProfile` are deliberately ABSENT, and the two lines above are exactly why
|
|
180
|
+
// this comment exists: their reasoning reads as though it should apply here too, and it must
|
|
181
|
+
// not. An image and a skills directory are toolchain; a resolved credential is a capability.
|
|
182
|
+
// A child's `task` is AGENT-AUTHORED (read off the request file below), so inheriting the
|
|
183
|
+
// binding would let a completed agent re-run itself against the operator's vault with a prompt
|
|
184
|
+
// it wrote for itself. The operator's grant was to the trigger they reviewed, not to whatever
|
|
185
|
+
// that job decides to queue next. A chained child that genuinely needs a secret gets it from a
|
|
186
|
+
// trigger of its own, which is an operator edit to a reviewed file.
|
|
179
187
|
chainDepth: childDepth,
|
|
180
188
|
parentJobId: job.id,
|
|
181
189
|
// chainedJobId deliberately does NOT take the image: the child's identity is (parent, flow, task).
|