@edgehero/pi-dispatch 1.2.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +11 -0
- package/deploy/docker-compose.yml +5 -1
- package/package.json +2 -1
- package/src/config.mjs +17 -0
- package/src/doctor.mjs +111 -4
- package/src/env-allowlist.mjs +31 -1
- package/src/forges.mjs +14 -0
- package/src/get-token.mjs +6 -1
- package/src/index.mjs +16 -0
- package/src/outbox.mjs +8 -0
- package/src/processor.mjs +141 -2
- package/src/queue.mjs +35 -3
- package/src/reserved-env.mjs +40 -0
- package/src/run-container.mjs +6 -2
- package/src/run-history.mjs +50 -31
- package/src/runtime-settings.mjs +40 -0
- package/src/schedules.mjs +1 -1
- package/src/secret-profiles.mjs +119 -0
- package/src/secrets.mjs +319 -0
- package/src/start.mjs +41 -3
- package/src/triggers-file.mjs +403 -0
- package/src/triggers.mjs +478 -21
package/src/secrets.mjs
ADDED
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Per-trigger secret references, resolved HOST-SIDE before a container starts (REQ-TRIGGER-SECRETS, #225).
|
|
3
|
+
*
|
|
4
|
+
* A trigger may carry `run.secrets`, a map of environment variable name to an OPAQUE REFERENCE, and
|
|
5
|
+
* `run.secretsProfile`, the name of one resolver the operator declared. The worker runs that resolver once
|
|
6
|
+
* per reference, takes its stdout as the value, and injects the values into the closed container env map
|
|
7
|
+
* exactly as the provider credential is injected. The job holds no manager credential, reaches no vault,
|
|
8
|
+
* and cannot enumerate one -- so `docs/secrets.md`'s rule stays literally true: what crosses the container
|
|
9
|
+
* boundary is a value, never the thing that can fetch values.
|
|
10
|
+
*
|
|
11
|
+
* THE REFERENCE GRAMMAR IS NOT OURS. `op://vault/item/field`, `secret/data/ci#stripe` and a bare name are
|
|
12
|
+
* all correct inputs, because what parses them is a two-line script the operator wrote. #206 refused
|
|
13
|
+
* "endorsing a vendor" and #209 refused "depending on, or blessing, any particular secrets manager -- the
|
|
14
|
+
* seam is a command". This module is that seam, moved from boot time (`DES-SERVICE-ENV-SETUP-SEAM`'s
|
|
15
|
+
* `--env-setup`) to job time, and it must never grow a regex that recognises one manager's notation.
|
|
16
|
+
*
|
|
17
|
+
* THE RESOLVER'S EXIT CODE IS THE RUNNER'S EXIT CODE. `INT-RUNNER-EXIT-CODE-PROTOCOL` is reused verbatim
|
|
18
|
+
* rather than invented here: 0 carries the value, 1 says "I could not reach my manager" and RETRIES, 2 says
|
|
19
|
+
* "that reference is wrong" and does not. Folding every nonzero exit into a refusal, which is the obvious
|
|
20
|
+
* design, breaks `CONST-RETRY-INFRA-ONLY` in the expensive direction: a vault unreachable for twenty
|
|
21
|
+
* seconds would permanently burn a delivery that BullMQ's `attempts: 2` would have recovered, and a webhook
|
|
22
|
+
* does not redeliver itself. `docs/secrets.md` already tells operators to ask what a manager does to their
|
|
23
|
+
* exit code, so the question is one they are pointed at rather than one this invents for them.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { spawn } from "node:child_process";
|
|
27
|
+
import { realpathSync, statSync } from "node:fs";
|
|
28
|
+
import { providerKeyVars } from "./env-allowlist.mjs";
|
|
29
|
+
import { EXIT_POLICY } from "./exit-code.mjs";
|
|
30
|
+
import { mergeSecretProfiles, withinRoots } from "./secret-profiles.mjs";
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* The largest value a resolver may print. `outbox.mjs` caps a request file at 4 KiB, which is too small
|
|
34
|
+
* here (a 4096-bit PEM chain, a service-account JSON -- this project already carries
|
|
35
|
+
* `GITHUB_APP_PRIVATE_KEY` as an env value), and `materialize.mjs` allows git 16 MiB, which per reference
|
|
36
|
+
* is a memory bomb. 64 KiB holds any real credential and bounds a full 16-reference job at 1 MiB.
|
|
37
|
+
*
|
|
38
|
+
* On overflow the child is killed and the job refuses. NEVER TRUNCATED: a silently shortened credential is
|
|
39
|
+
* the exact failure this gate exists to prevent, and it is the direction `execFile`'s own `maxBuffer` takes.
|
|
40
|
+
*/
|
|
41
|
+
export const MAX_SECRET_BYTES = 64 * 1024;
|
|
42
|
+
|
|
43
|
+
/** SIGTERM, then this long, then SIGKILL. doctor's capture helper does a bare kill() with no escalation, */
|
|
44
|
+
/** which is fine for a diagnostic and not for something holding a pipe in front of a paid container. */
|
|
45
|
+
const KILL_GRACE_MS = 2000;
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* The profile a trigger selects when it names none. A single-manager deployment declares one profile
|
|
49
|
+
* called `default` and never writes the field.
|
|
50
|
+
*
|
|
51
|
+
* A LOOKUP KEY, never a fallback: there is deliberately no "if only one profile is declared, use it" rule.
|
|
52
|
+
* `env-allowlist.mjs` records what that costs -- its forge lookup was once an `if gitlab / else github`,
|
|
53
|
+
* and the `else` handed a credential to the wrong host. A rule whose meaning changes the day an operator
|
|
54
|
+
* declares a second profile is that mistake with a slower fuse.
|
|
55
|
+
*/
|
|
56
|
+
export const DEFAULT_SECRETS_PROFILE = "default";
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Whether this job asked for secrets at all.
|
|
60
|
+
*
|
|
61
|
+
* EITHER field arms it. `secretsProfile` alone cannot reach a wired worker (parseTriggers refuses a profile
|
|
62
|
+
* that resolves nothing), but this predicate also guards the processor's fail-closed default, which runs
|
|
63
|
+
* under wirings that never saw the validator -- a bare `makeProcessor` in a test, or a job whose data was
|
|
64
|
+
* queued by a different version. Treating a lone profile as unarmed there would silently start a job the
|
|
65
|
+
* operator believes is bound to a vault.
|
|
66
|
+
*
|
|
67
|
+
* Exported so the gate and its default share ONE definition of "armed", which is the discipline
|
|
68
|
+
* `processor.mjs` already keeps for `run.resume` (`=== true`, spelled the same in the gate and in
|
|
69
|
+
* prepare-github, "so the gate and the feature cannot disagree about what armed means").
|
|
70
|
+
*/
|
|
71
|
+
export function secretsArmed(job) {
|
|
72
|
+
if (typeof job?.secretsProfile === "string" && job.secretsProfile !== "") return true;
|
|
73
|
+
const secrets = job?.secrets;
|
|
74
|
+
return secrets !== null && typeof secrets === "object" && Object.keys(secrets).length > 0;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Run one resolver against one reference. Resolves a discriminated result; never throws, never rejects.
|
|
79
|
+
*
|
|
80
|
+
* `failure` is OUR vocabulary, never the resolver's words:
|
|
81
|
+
* - `policy` the reference is wrong, absent or denied (exit 2), or the output was empty, oversized,
|
|
82
|
+
* or carried a NUL. Determinate: retrying cannot change the answer.
|
|
83
|
+
* - `infra` the manager could not be reached (exit 1), an unrecognised exit, a spawn fault, a timeout.
|
|
84
|
+
*
|
|
85
|
+
* An unrecognised exit code is INFRA on `decideRetry`'s own reasoning: a runner that exits with something
|
|
86
|
+
* we do not recognise is one we cannot reason about, and retrying-then-alerting beats accepting it as done.
|
|
87
|
+
*/
|
|
88
|
+
function runResolver(spawnFn, path, reference, { timeoutMs, hostEnv, signal }) {
|
|
89
|
+
return new Promise((resolve) => {
|
|
90
|
+
let child;
|
|
91
|
+
try {
|
|
92
|
+
child = spawnFn(path, [reference], {
|
|
93
|
+
// stdin IGNORED, so a resolver that prompts (`op signin`, `vault login`) dies at once rather
|
|
94
|
+
// than blocking to the timeout behind an invisible password prompt on a headless host.
|
|
95
|
+
//
|
|
96
|
+
// stdout and stderr are SEPARATE pipes. doctor's capture helper merges them, and its stated
|
|
97
|
+
// reason (gh moves human output between the two) does not survive here: a merge splices a
|
|
98
|
+
// deprecation warning into the secret, and because the value is printed nowhere the corruption
|
|
99
|
+
// stays invisible until the container authenticates with a wrong string and the agent writes a
|
|
100
|
+
// plausible report about why the integration is down.
|
|
101
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
102
|
+
// The worker's own environment, verbatim. The resolver has to authenticate to the manager, and
|
|
103
|
+
// the whole architecture of docs/secrets.md is that its credential lives here. Narrowing would
|
|
104
|
+
// mean naming each vendor's variables, which is blessing vendors, and would be theatre anyway:
|
|
105
|
+
// a script running as the worker user can read /proc/self/environ. What it does NOT see is a
|
|
106
|
+
// forge token, and that is free rather than arranged -- this runs before the mint.
|
|
107
|
+
env: hostEnv,
|
|
108
|
+
// An ARRAY and never a shell string: no interpolation, no injection. docker-run.mjs states the
|
|
109
|
+
// rule for the only other place this project spawns on the job path.
|
|
110
|
+
shell: false,
|
|
111
|
+
});
|
|
112
|
+
} catch {
|
|
113
|
+
resolve({ value: null, failure: "infra", detail: "spawn", code: null, stderrBytes: 0 });
|
|
114
|
+
return;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const chunks = [];
|
|
118
|
+
let bytes = 0;
|
|
119
|
+
let stderrBytes = 0;
|
|
120
|
+
let done = false;
|
|
121
|
+
let overflowed = false;
|
|
122
|
+
let killTimer = null;
|
|
123
|
+
|
|
124
|
+
const finish = (result) => {
|
|
125
|
+
if (done) return;
|
|
126
|
+
done = true;
|
|
127
|
+
clearTimeout(timer);
|
|
128
|
+
signal?.removeEventListener?.("abort", onAbort);
|
|
129
|
+
resolve(result);
|
|
130
|
+
};
|
|
131
|
+
// SIGTERM, then SIGKILL after a grace. The kill timer is deliberately NOT cleared by `finish`: the
|
|
132
|
+
// promise settles now, and the escalation still has to land on a child that ignored the first signal.
|
|
133
|
+
const stop = (why) => {
|
|
134
|
+
try {
|
|
135
|
+
child.kill();
|
|
136
|
+
} catch {}
|
|
137
|
+
killTimer = setTimeout(() => {
|
|
138
|
+
try {
|
|
139
|
+
child.kill("SIGKILL");
|
|
140
|
+
} catch {}
|
|
141
|
+
}, KILL_GRACE_MS);
|
|
142
|
+
killTimer.unref?.();
|
|
143
|
+
finish({ value: null, failure: why === "overflow" ? "policy" : "infra", detail: why, code: null, stderrBytes });
|
|
144
|
+
};
|
|
145
|
+
|
|
146
|
+
const timer = setTimeout(() => stop("timeout"), timeoutMs);
|
|
147
|
+
const onAbort = () => stop("aborted");
|
|
148
|
+
signal?.addEventListener?.("abort", onAbort, { once: true });
|
|
149
|
+
|
|
150
|
+
child.stdout?.on("data", (d) => {
|
|
151
|
+
bytes += d.length;
|
|
152
|
+
if (bytes > MAX_SECRET_BYTES) {
|
|
153
|
+
overflowed = true;
|
|
154
|
+
stop("overflow");
|
|
155
|
+
return;
|
|
156
|
+
}
|
|
157
|
+
chunks.push(d);
|
|
158
|
+
});
|
|
159
|
+
// A COUNTER, never a string. Keeping the bytes would leave the resolver's own words one refactor away
|
|
160
|
+
// from a public issue comment, and a resolver's stderr can echo the reference, the vault path, or in a
|
|
161
|
+
// careless tool the value itself. A count still tells an operator to go run it by hand.
|
|
162
|
+
child.stderr?.on("data", (d) => (stderrBytes += d.length));
|
|
163
|
+
child.on("error", () => finish({ value: null, failure: "infra", detail: "spawn", code: null, stderrBytes }));
|
|
164
|
+
child.on("close", (code) => {
|
|
165
|
+
if (overflowed || done) return;
|
|
166
|
+
if (code === EXIT_POLICY) return finish({ value: null, failure: "policy", detail: "exit", code, stderrBytes });
|
|
167
|
+
if (code !== 0) return finish({ value: null, failure: "infra", detail: "exit", code, stderrBytes });
|
|
168
|
+
const value = stripOneNewline(Buffer.concat(chunks).toString("utf8"));
|
|
169
|
+
// Exit 0 printing nothing is the silent-no-op class: a vault CLI that says nothing and succeeds
|
|
170
|
+
// would otherwise bind an empty variable and let the job run as though it were configured.
|
|
171
|
+
if (value === "") return finish({ value: null, failure: "policy", detail: "empty", code, stderrBytes });
|
|
172
|
+
// `-e NAME=VALUE` is an execve argv element and truncates at a NUL, silently handing the container
|
|
173
|
+
// a shortened credential. Interior NEWLINES are fine and must stay fine: it is an argv, not a
|
|
174
|
+
// shell string, and a PEM has them.
|
|
175
|
+
if (/\u0000/.test(value)) return finish({ value: null, failure: "policy", detail: "nul", code, stderrBytes });
|
|
176
|
+
finish({ value, failure: null, detail: null, code, stderrBytes });
|
|
177
|
+
});
|
|
178
|
+
});
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Strip EXACTLY ONE trailing newline, and a `\r` before it only if that strip happened.
|
|
183
|
+
*
|
|
184
|
+
* Every CLI in this space terminates with one newline (`vault kv get -field=`, `pass show`, `gcloud secrets
|
|
185
|
+
* versions access`), and leaving it in puts a newline inside an HTTP header value: some clients reject it,
|
|
186
|
+
* some send it, and the failure names nothing. `op read --no-newline` prints none and must keep working,
|
|
187
|
+
* which is why this is idempotent rather than a loop.
|
|
188
|
+
*
|
|
189
|
+
* NOT `.trim()`, in either direction. A passphrase may legitimately begin or end with a space, and a PEM
|
|
190
|
+
* ends with a newline whose removal must not cascade into the body. Silently mangling a credential is the
|
|
191
|
+
* same class of harm as splicing stderr into it.
|
|
192
|
+
*/
|
|
193
|
+
function stripOneNewline(text) {
|
|
194
|
+
if (text.endsWith("\r\n")) return text.slice(0, -2);
|
|
195
|
+
if (text.endsWith("\n")) return text.slice(0, -1);
|
|
196
|
+
return text;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Build the pre-spend resolver the processor injects.
|
|
201
|
+
*
|
|
202
|
+
* `(job) => { ok, secrets } | { profileUnknown } | { ambiguous } | { unresolved, ... } | { unreachable, ... }`
|
|
203
|
+
*
|
|
204
|
+
* `envProfiles` is parsed at BOOT by `config.mjs`; `overlayProfiles` arrives per job, because the settings
|
|
205
|
+
* overlay is read at each job start (INT-CONFIG-OVERLAY-CONTRACT) and an operator who declares a profile in
|
|
206
|
+
* the panel should not have to restart the worker. The processor never reads a file: the table arrives
|
|
207
|
+
* already parsed, exactly as `sessionsDir` does.
|
|
208
|
+
*
|
|
209
|
+
* SEQUENTIAL, in the operator's own writing order, deduped by unique reference. Not throughput: with two
|
|
210
|
+
* broken references `Promise.all` refuses on whichever loses the race, so one broken trigger would blame a
|
|
211
|
+
* different variable on different runs. Sequential always names the first, so an operator fixes them in the
|
|
212
|
+
* order they wrote them. It also caps the blast radius of a refusal at the references BEFORE the failure,
|
|
213
|
+
* rather than pulling all sixteen live credentials into memory for a job that will never run, and spares
|
|
214
|
+
* every managed vault a burst of N x PI_CONCURRENCY parallel reads nobody sized for. The cost is nil: this
|
|
215
|
+
* band is followed immediately by a git clone measured in tens of seconds.
|
|
216
|
+
*
|
|
217
|
+
* NOTHING IS ZEROED, and saying so is the honest half. JS strings are immutable and unzeroable; pretending
|
|
218
|
+
* otherwise would be theatre. What IS true: the returned map is a block-scoped local in `runJob`, never
|
|
219
|
+
* attached to `job` or `prepared`, never returned in a result, never passed to `log`, `comment` or
|
|
220
|
+
* `recordRun` -- and `buildRecord` is an explicit literal with no spread, so it cannot pick it up by
|
|
221
|
+
* accident. That property is load-bearing now, not incidental.
|
|
222
|
+
*/
|
|
223
|
+
export function makeSecretsResolver({
|
|
224
|
+
envProfiles = {},
|
|
225
|
+
roots = [],
|
|
226
|
+
// The operator's PI_FORWARD_ENV names. Needed here rather than in triggers.mjs because that validator
|
|
227
|
+
// is env-free: see the reserved-name check below.
|
|
228
|
+
forwardEnv = [],
|
|
229
|
+
timeoutMs = 10000,
|
|
230
|
+
spawnFn = spawn,
|
|
231
|
+
hostEnv = process.env,
|
|
232
|
+
// Injected so the executable probe is testable without a real script on disk. `statSync`, NOT `lstat`,
|
|
233
|
+
// which is a deliberate divergence from processor.mjs's isReadableDir: that one judges a symlinked
|
|
234
|
+
// DIRECTORY on its own inode, while this must answer the same question `spawn` will ask, and `spawn`
|
|
235
|
+
// follows symlinks -- so `lstat` would refuse an ordinary /usr/local/bin/op -> ../Cellar/... The mode is
|
|
236
|
+
// deliberately NOT checked here: a group- or world-writable resolver is a doctor WARNING, on the
|
|
237
|
+
// env-setup precedent, because refusing a job over a mode bit at job time is a new class of refusal.
|
|
238
|
+
realExecutablePath = (p) => {
|
|
239
|
+
const real = realpathSync(p);
|
|
240
|
+
const st = statSync(real);
|
|
241
|
+
return st.isFile() && (st.mode & 0o111) !== 0 ? real : null;
|
|
242
|
+
},
|
|
243
|
+
log = () => {},
|
|
244
|
+
} = {}) {
|
|
245
|
+
return async function resolveSecrets(job, { signal, overlayProfiles = {} } = {}) {
|
|
246
|
+
const merged = mergeSecretProfiles(envProfiles, overlayProfiles);
|
|
247
|
+
if (merged.ambiguous) return { ambiguous: merged.ambiguous };
|
|
248
|
+
|
|
249
|
+
const wanted = typeof job?.secretsProfile === "string" && job.secretsProfile !== "" ? job.secretsProfile : DEFAULT_SECRETS_PROFILE;
|
|
250
|
+
const declared = merged.profiles[wanted];
|
|
251
|
+
// A LOOKUP, never a fallback. There is deliberately no "if only one profile is declared, use it"
|
|
252
|
+
// rule: env-allowlist.mjs records what the last `else` of that shape cost, and a default whose
|
|
253
|
+
// meaning changes the day a second profile appears is that mistake with a slower fuse.
|
|
254
|
+
if (typeof declared !== "string") return { profileUnknown: wanted };
|
|
255
|
+
|
|
256
|
+
let path;
|
|
257
|
+
try {
|
|
258
|
+
path = realExecutablePath(declared);
|
|
259
|
+
} catch {
|
|
260
|
+
path = null;
|
|
261
|
+
}
|
|
262
|
+
if (!path) return { profileUnknown: wanted };
|
|
263
|
+
// The bound is checked on the REALPATH, and in the WORKER rather than only where a profile is
|
|
264
|
+
// written. A panel-side check alone would be cosmetic: the settings overlay defaults into the OS
|
|
265
|
+
// temp directory, so on a multi-user host the directory can be pre-created by anyone. Checking here
|
|
266
|
+
// caps a tampered overlay at "choose among scripts the operator allowlisted" instead of "name any
|
|
267
|
+
// executable on the host". An env-declared profile is exempt: PI_SECRET_PROFILES already lives
|
|
268
|
+
// beside the forge tokens and the App key, so requiring roots for it would bound nothing and would
|
|
269
|
+
// break every deployment that never wanted panel authoring.
|
|
270
|
+
const fromOverlay = !(wanted in envProfiles);
|
|
271
|
+
if (fromOverlay && !withinRoots(path, roots)) return { profileUnknown: wanted };
|
|
272
|
+
|
|
273
|
+
const references = job?.secrets ?? {};
|
|
274
|
+
|
|
275
|
+
// The half of the reserved-name refusal that parseTriggers structurally cannot make. It refuses every
|
|
276
|
+
// STATICALLY knowable name (the mint's, each forge's host var, the worker-only secrets, the egress
|
|
277
|
+
// policy's, and the closed map's own PI_*/PLAYWRIGHT_*), but two more sets are DEPLOYMENT STATE:
|
|
278
|
+
//
|
|
279
|
+
// - the provider credential's variable names, which come from `findEnvKeys(provider, hostEnv)` and so
|
|
280
|
+
// depend on the job's resolved provider AND on what this host has set;
|
|
281
|
+
// - PI_FORWARD_ENV, which is an operator env list.
|
|
282
|
+
//
|
|
283
|
+
// Both are load-bearing rather than tidy. buildContainerEnv writes the provider credential BEFORE this
|
|
284
|
+
// feature's assignment, so a trigger binding ANTHROPIC_API_KEY would silently replace the credential
|
|
285
|
+
// every job of that trigger spends -- verbatim the failure env-allowlist.mjs's header exists to
|
|
286
|
+
// prevent, arriving through a new door. PI_FORWARD_ENV is a second route to the same place, because
|
|
287
|
+
// that list is how a CUSTOM provider's key reaches the container.
|
|
288
|
+
//
|
|
289
|
+
// Refused rather than resolved by ordering, which is the division of labour the mint already keeps:
|
|
290
|
+
// ordering is the backstop, the refusal is the gate.
|
|
291
|
+
const reserved = new Set([...(providerKeyVars(job?.provider, hostEnv) ?? []), ...forwardEnv]);
|
|
292
|
+
for (const name of Object.keys(references)) {
|
|
293
|
+
if (reserved.has(name)) return { reserved: name };
|
|
294
|
+
}
|
|
295
|
+
const secrets = {};
|
|
296
|
+
const seen = new Map();
|
|
297
|
+
for (const [name, reference] of Object.entries(references)) {
|
|
298
|
+
if (seen.has(reference)) {
|
|
299
|
+
// Two names sharing one reference resolve once, so they are guaranteed identical rather than
|
|
300
|
+
// straddling a rotation that lands between two reads.
|
|
301
|
+
secrets[name] = seen.get(reference);
|
|
302
|
+
continue;
|
|
303
|
+
}
|
|
304
|
+
const outcome = await runResolver(spawnFn, path, reference, { timeoutMs, hostEnv, signal });
|
|
305
|
+
if (outcome.failure) {
|
|
306
|
+
// The NAME, our own failure vocabulary, the script's small integer exit and a byte COUNT.
|
|
307
|
+
// Never the resolver's words, never the reference, never the path (no-pii-in-logs).
|
|
308
|
+
log("secret_resolve_failed", { name, failure: outcome.failure, detail: outcome.detail, code: outcome.code ?? null, stderrBytes: outcome.stderrBytes });
|
|
309
|
+
const shape = { failure: outcome.detail, code: outcome.code ?? null, stderrBytes: outcome.stderrBytes };
|
|
310
|
+
// The whole job refuses, never a partial set: a job missing one of three secrets fails deep
|
|
311
|
+
// inside a paid container in a way that reads as a flow bug.
|
|
312
|
+
return outcome.failure === "infra" ? { unreachable: name, ...shape } : { unresolved: name, ...shape };
|
|
313
|
+
}
|
|
314
|
+
seen.set(reference, outcome.value);
|
|
315
|
+
secrets[name] = outcome.value;
|
|
316
|
+
}
|
|
317
|
+
return { ok: true, secrets };
|
|
318
|
+
};
|
|
319
|
+
}
|
package/src/start.mjs
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { execFile } from "node:child_process";
|
|
2
2
|
import { watch } from "node:fs";
|
|
3
|
-
import { dirname, basename } from "node:path";
|
|
3
|
+
import { dirname, basename, join } from "node:path";
|
|
4
4
|
import { promisify } from "node:util";
|
|
5
5
|
import { configError, loadConfig } from "./config.mjs";
|
|
6
6
|
import { makeRedisClient, parseConnection } from "./connection.mjs";
|
|
@@ -22,9 +22,11 @@ import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare
|
|
|
22
22
|
import { listRunningSandboxes } from "./sandbox.mjs";
|
|
23
23
|
import { makeSandboxReaper } from "./sandbox-store.mjs";
|
|
24
24
|
import { makeSessionStore } from "./session-store.mjs";
|
|
25
|
+
import { makeCheckOnceSpent, makeDisarmOnce } from "./triggers-file.mjs";
|
|
25
26
|
import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
|
|
26
27
|
import { makeQueue } from "./queue.mjs";
|
|
27
28
|
import { makeRunContainer } from "./run-container.mjs";
|
|
29
|
+
import { makeSecretsResolver } from "./secrets.mjs";
|
|
28
30
|
import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
|
|
29
31
|
import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
|
|
30
32
|
import { loadSchedules } from "./schedules.mjs";
|
|
@@ -162,6 +164,7 @@ export async function startWorker(
|
|
|
162
164
|
makeLogReaper: makeLogReaperFn = makeLogReaper,
|
|
163
165
|
makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
|
|
164
166
|
makeRunContainer: makeRunContainerFn = makeRunContainer,
|
|
167
|
+
makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
|
|
165
168
|
makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
|
|
166
169
|
makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
|
|
167
170
|
makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
|
|
@@ -311,8 +314,25 @@ export async function startWorker(
|
|
|
311
314
|
} catch (err) {
|
|
312
315
|
log("session_reaper_skipped", { reason: err?.message });
|
|
313
316
|
}
|
|
314
|
-
|
|
317
|
+
// The one-shot file path (issue #231): PI_TRIGGERS_FILE, else ./triggers.json against this process's
|
|
318
|
+
// cwd -- doctor's own fallback, chosen for doctor's own reason ("the two must read the same file"),
|
|
319
|
+
// and deliberately NOT config.triggersFile, whose null means "cron disabled" and must keep meaning
|
|
320
|
+
// that: under that knob the DEFAULT single-host deployment would have a firing receiver and a worker
|
|
321
|
+
// that can neither disarm nor pre-spend-check.
|
|
322
|
+
const onceTriggersFile = env.PI_TRIGGERS_FILE ?? join(process.cwd(), "triggers.json");
|
|
323
|
+
const disarmOnce = makeDisarmOnce({ triggersPath: onceTriggersFile, log });
|
|
324
|
+
const recordRun = ({ job, result, error, startedAt, endedAt }) => {
|
|
315
325
|
writeRecord(buildRecord({ job, result, error, startedAt, endedAt }));
|
|
326
|
+
// Strictly AFTER the durable record: "fired" means "produced a run record", and the crash
|
|
327
|
+
// direction this ordering buys is the chosen one -- an armed one-shot with a record, never a
|
|
328
|
+
// disarm before writeRecord RETURNED. Returned, not succeeded: the record writer swallows fs
|
|
329
|
+
// errors by contract (run_record_failed), so a full disk still spends the one-shot -- the
|
|
330
|
+
// alternative, skipping the disarm on a failed record write, would re-fire it unbounded. Fire-and-forget: the hook never rejects, and the record path must not
|
|
331
|
+
// wait on a lock retry. An uncontended disarm completes synchronously inside this call; the
|
|
332
|
+
// one loss window is a drain's process.exit landing mid-lock-retry sleep, which loses only
|
|
333
|
+
// the disarm -- the same chosen direction, met at shutdown instead of a crash.
|
|
334
|
+
void disarmOnce({ job, endedAt });
|
|
335
|
+
};
|
|
316
336
|
|
|
317
337
|
// INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
|
|
318
338
|
// closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the eight
|
|
@@ -326,7 +346,12 @@ export async function startWorker(
|
|
|
326
346
|
log("settings_overlay_invalid", { reason: res.invalid, settingsFile });
|
|
327
347
|
return { invalid: res.invalid };
|
|
328
348
|
}
|
|
329
|
-
|
|
349
|
+
// REQ-TRIGGER-SECRETS rides ALONGSIDE the ten tunables rather than inside them. `effectiveSettings`
|
|
350
|
+
// resolves `overlay > env` over a fixed ten-key literal, and its own tests pin that key set and
|
|
351
|
+
// assert an empty overlay returns the config verbatim -- so an eleventh key there would break both,
|
|
352
|
+
// and would also claim a precedence this key deliberately does not have (a name declared in both
|
|
353
|
+
// sources is refused per delivery, not silently won by either).
|
|
354
|
+
return { ...effectiveSettings(config, res.overlay), secretProfiles: res.overlay?.secretProfiles ?? {} };
|
|
330
355
|
};
|
|
331
356
|
|
|
332
357
|
// Resolve the Worker constructor's slot count once from the overlay: a present overlay may raise or lower
|
|
@@ -399,6 +424,11 @@ export async function startWorker(
|
|
|
399
424
|
pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
|
|
400
425
|
deps: {
|
|
401
426
|
collectChain,
|
|
427
|
+
// The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
|
|
428
|
+
// only on a FOREIGN positive mark (index.mjs binds the real queue jobId so a retry of the
|
|
429
|
+
// spending delivery is excused). In the compose topology this check is the once-enforcement
|
|
430
|
+
// layer, because the receiver's single-file :ro mount pins a dead inode until restart.
|
|
431
|
+
checkOnceSpent: makeCheckOnceSpent({ triggersPath: onceTriggersFile }),
|
|
402
432
|
// One deployment default, two consumers, adjacent by construction: the preflight that refuses a missing
|
|
403
433
|
// image BEFORE the budget slot, and the factory that puts it in the argv. Both resolve a trigger's own
|
|
404
434
|
// `run.image` through the same resolveJobImage, so the image that was checked is the image that runs.
|
|
@@ -421,6 +451,14 @@ export async function startWorker(
|
|
|
421
451
|
// real path; the difference shows under an injected env, where the store would be built from the
|
|
422
452
|
// synthetic value while the gate read the process one.
|
|
423
453
|
sessionsDir: config.sessionsDir,
|
|
454
|
+
// REQ-TRIGGER-SECRETS. Built here for the image and egress preflights' reason: one deployment
|
|
455
|
+
// value, one place, so the gate that refuses an unknown profile and the spawn that runs it cannot
|
|
456
|
+
// disagree about which resolvers exist. The env-declared table is parsed once at boot (it is env,
|
|
457
|
+
// and a change to it is a restart), while the OVERLAY half arrives per job through index.mjs,
|
|
458
|
+
// because the settings file is read at each job start and an operator who declares a profile in
|
|
459
|
+
// the panel should not have to restart the worker to use it. A deployment that declares nothing
|
|
460
|
+
// spawns nothing at all: the gate only calls this when a trigger is armed.
|
|
461
|
+
resolveSecrets: makeSecretsResolverFn({ envProfiles: config.secretProfiles, roots: config.secretResolverRoots, timeoutMs: config.secretResolveTimeoutMs, forwardEnv: config.forwardEnv, log }),
|
|
424
462
|
runContainer: makeRunContainerFn({
|
|
425
463
|
image: config.jobImage,
|
|
426
464
|
hostEnv: env,
|