@edgehero/pi-dispatch 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +26 -1
- package/deploy/docker-compose.yml +59 -0
- package/deploy/egress-proxy.conf +36 -0
- package/deploy/receiver.service +11 -0
- package/deploy/worker-env-wrapper.cmd +41 -4
- package/deploy/worker-env-wrapper.sh +96 -7
- package/package.json +1 -1
- package/src/azure-prompt.mjs +57 -9
- package/src/config.mjs +133 -6
- package/src/docker-run.mjs +16 -1
- package/src/doctor.mjs +398 -13
- package/src/egress.mjs +221 -0
- package/src/env-allowlist.mjs +16 -1
- package/src/forgejo-prompt.mjs +65 -11
- package/src/get-token.mjs +5 -3
- package/src/github-prompt.mjs +11 -2
- package/src/gitlab-prompt.mjs +65 -11
- package/src/init.mjs +30 -0
- package/src/packages.mjs +4 -1
- package/src/prepare-github.mjs +6 -4
- package/src/processor.mjs +67 -5
- package/src/run-container.mjs +42 -6
- package/src/run-history.mjs +41 -3
- package/src/sandbox-cli.mjs +24 -3
- package/src/sandbox.mjs +14 -1
- package/src/service.mjs +289 -25
- package/src/session-store.mjs +294 -15
- package/src/start.mjs +29 -0
- package/src/triggers.mjs +11 -12
- package/src/up.mjs +58 -0
package/src/session-store.mjs
CHANGED
|
@@ -44,6 +44,54 @@ import { sessionKeyFor } from "./session-key.mjs";
|
|
|
44
44
|
export const SESSION_FILE_NAME = "current.jsonl";
|
|
45
45
|
const PI_VERSION_FILE = "pi-version";
|
|
46
46
|
const LOCK_FILE = "lock";
|
|
47
|
+
/**
|
|
48
|
+
* How many times in a row this key's transcript has been HANDED TO A CONTAINER. A counter rather than a
|
|
49
|
+
* derivation, because there is nothing to derive it from: the run record deliberately carries no session
|
|
50
|
+
* key (DES-SESSION-KEY-IS-DERIVED-NOT-INDEXED), so counting past runs would need the key->record index
|
|
51
|
+
* that entry refuses. One integer beside the transcript it describes is not that index; it is keyed state
|
|
52
|
+
* written where the key already is, and it answers exactly one question rather than being a query
|
|
53
|
+
* surface. Maintained even when no bound is set, deliberately -- see the write in promoteSession.
|
|
54
|
+
*
|
|
55
|
+
* IT COUNTS THE HOST'S DELIVERIES, NOT PI'S CONTINUATIONS, and that is the whole security of this bound.
|
|
56
|
+
* It counted pi's `resumed` first, on the reasoning that a transcript pi declined to continue extended
|
|
57
|
+
* nothing. That reasoning is wrong here, because the agent owns /session and therefore chooses what pi
|
|
58
|
+
* makes of the file: a transcript carrying a valid header and payload on lines pi's parser DROPS is
|
|
59
|
+
* delivered by the host every run while pi reports zero messages, so the counter reset every run and the
|
|
60
|
+
* chain bound never fired -- measured, not theorised. The host's own decision to hand the file over is
|
|
61
|
+
* the one fact in this exchange that nothing inside the container can influence.
|
|
62
|
+
*/
|
|
63
|
+
const RESUME_CHAIN_FILE = "resume-chain";
|
|
64
|
+
/**
|
|
65
|
+
* How full the context was when the run that wrote this transcript ended, as `<tokens> <window>`. Both
|
|
66
|
+
* numbers, not a precomputed percentage: the denominator is what makes the numerator readable later, and
|
|
67
|
+
* an operator looking at a refusal should be able to see what it was judged against.
|
|
68
|
+
*
|
|
69
|
+
* Reported BY THE CONTAINER, which is the only place the number exists: pi computes it from the session
|
|
70
|
+
* it is holding. That puts it at the same trust level as `turns` and `tokens`, and the residual is
|
|
71
|
+
* recorded in OQ-003 rather than papered over -- there is no host-side alternative that is not equally
|
|
72
|
+
* agent-influenced, since the transcript itself is agent-written.
|
|
73
|
+
*/
|
|
74
|
+
const CONTEXT_FILE = "context";
|
|
75
|
+
/** Both sidecar formats are a handful of bytes. Generous, and still nowhere near a job's wall clock. */
|
|
76
|
+
const SIDECAR_MAX_BYTES = 4096;
|
|
77
|
+
/**
|
|
78
|
+
* The host-effective provider and model as one token, or null when the job names neither.
|
|
79
|
+
*
|
|
80
|
+
* CONSERVATIVE BY CONSTRUCTION: the sidecar is whitespace-delimited, so a value carrying a space would
|
|
81
|
+
* split the record and be read back as a different field. Rather than escape, refuse: anything outside
|
|
82
|
+
* the charset the run record already validates model ids against is no identity, and no identity means
|
|
83
|
+
* the reading stays usable rather than being thrown away.
|
|
84
|
+
*/
|
|
85
|
+
function modelIdentity(job) {
|
|
86
|
+
const provider = typeof job?.provider === "string" ? job.provider : "";
|
|
87
|
+
const model = typeof job?.model === "string" ? job.model : "";
|
|
88
|
+
if (provider === "" || model === "") return null;
|
|
89
|
+
// Lowercased first, the same normalisation the run record's own model ids get, so a trigger written
|
|
90
|
+
// `Claude-Sonnet` and one written `claude-sonnet` are one model rather than two -- and so that a
|
|
91
|
+
// perfectly ordinary id does not fall out of the charset below and silently stop stamping.
|
|
92
|
+
const id = `${provider}/${model}`.toLowerCase();
|
|
93
|
+
return /^[a-z0-9][a-z0-9._:/-]{0,127}$/.test(id) ? id : null;
|
|
94
|
+
}
|
|
47
95
|
|
|
48
96
|
/**
|
|
49
97
|
* Read-path outcomes. Every one is a named cold start rather than a bare `false`: a feature that fails
|
|
@@ -56,6 +104,9 @@ export function makeSessionStore({
|
|
|
56
104
|
sessionsDir,
|
|
57
105
|
ttlDays,
|
|
58
106
|
maxBytes,
|
|
107
|
+
maxAgeDays = 0,
|
|
108
|
+
maxResumeChain = 0,
|
|
109
|
+
maxContextPct = null,
|
|
59
110
|
log = () => {},
|
|
60
111
|
now = () => Date.now(),
|
|
61
112
|
fs = { copyFileSync, lstatSync, mkdirSync, openSync, closeSync, readFileSync, readdirSync, renameSync, rmSync, unlinkSync, writeFileSync },
|
|
@@ -83,11 +134,16 @@ export function makeSessionStore({
|
|
|
83
134
|
// CLI run, an unresolvable head ref), so it gets no mount and no transcript on disk.
|
|
84
135
|
if (key === null) return null;
|
|
85
136
|
|
|
137
|
+
// The model this job will actually run, for the context bound. A key is (kind, repo, ref) and
|
|
138
|
+
// carries NO model, so two triggers on one issue can name different ones, and the same token
|
|
139
|
+
// count is 78% of a 32k window and 2.5% of a 1M one. Carried on the session object rather than
|
|
140
|
+
// read again at promote time, so the reading is stamped with the model that produced it.
|
|
141
|
+
const modelId = modelIdentity(job);
|
|
86
142
|
const hostDir = join(jobDir, "session");
|
|
87
143
|
const staged = join(hostDir, SESSION_FILE_NAME);
|
|
88
144
|
fs.mkdirSync(hostDir, { recursive: true, mode: 0o700 });
|
|
89
145
|
|
|
90
|
-
const verdict = readCanonical(key, piVersion);
|
|
146
|
+
const verdict = readCanonical(key, piVersion, modelId);
|
|
91
147
|
if (verdict.resume) {
|
|
92
148
|
fs.copyFileSync(canonicalFile(key), staged);
|
|
93
149
|
} else {
|
|
@@ -98,7 +154,7 @@ export function makeSessionStore({
|
|
|
98
154
|
fs.writeFileSync(staged, "");
|
|
99
155
|
}
|
|
100
156
|
log("session_resolved", { key, resume: verdict.resume, reason: verdict.reason });
|
|
101
|
-
return { hostDir, key, ...verdict };
|
|
157
|
+
return { hostDir, key, modelId, ...verdict };
|
|
102
158
|
} catch (err) {
|
|
103
159
|
// A history fault must never fail the prepare that asked.
|
|
104
160
|
log("session_store_failed", { phase: "resolve", reason: err?.message });
|
|
@@ -114,7 +170,7 @@ export function makeSessionStore({
|
|
|
114
170
|
* one key is a real shape (REQ-QUEUE-BURST-NO-DROP), and last-write-wins there would interleave two
|
|
115
171
|
* agents' turns into one transcript.
|
|
116
172
|
*/
|
|
117
|
-
function promoteSession(session, { piVersion = null } = {}) {
|
|
173
|
+
function promoteSession(session, { piVersion = null, context = null } = {}) {
|
|
118
174
|
// The second DI-seam backstop, and unreachable for the same reason as the `!sessionsDir` return
|
|
119
175
|
// above: sessionKeyFor is total and binary (null, or 32 hex chars), so resolveSession returns null
|
|
120
176
|
// rather than a keyless session, and processor.mjs only calls this when prepare handed it one. Kept
|
|
@@ -137,16 +193,42 @@ export function makeSessionStore({
|
|
|
137
193
|
let fd;
|
|
138
194
|
try {
|
|
139
195
|
fd = fs.openSync(lock, "wx"); // exclusive create IS the lock; no daemon, no lease
|
|
140
|
-
} catch {
|
|
196
|
+
} catch (err) {
|
|
197
|
+
// EEXIST is the only failure that MEANS locked. A read-only directory, a full disk or a
|
|
198
|
+
// vanished store all failed to create the lock too, and reporting those as `locked` sends an
|
|
199
|
+
// operator looking for a stuck lock file that does not exist. Anything else falls through to
|
|
200
|
+
// the outer catch and reports `promote-failed`, which is what actually happened.
|
|
201
|
+
if (err?.code !== "EEXIST") throw err;
|
|
141
202
|
log("session_promote_skipped", { key: session.key, reason: "locked" });
|
|
142
203
|
return { promoted: false, reason: "locked" };
|
|
143
204
|
}
|
|
144
205
|
try {
|
|
145
206
|
// Atomic swap: a reader either sees the old file or the new one, never a half-written one.
|
|
146
207
|
const tmp = `${canonicalFile(session.key)}.incoming`;
|
|
208
|
+
try {
|
|
209
|
+
// `copyFileSync` follows a link at the DESTINATION, so a link planted at this name would
|
|
210
|
+
// receive the whole transcript and leave the canonical path pointing at it. The key
|
|
211
|
+
// directory's name is derived rather than random, so the path is precomputable by anyone
|
|
212
|
+
// who knows the repository and the branch; unlinking removes the link, never its target.
|
|
213
|
+
fs.unlinkSync(tmp);
|
|
214
|
+
} catch {
|
|
215
|
+
// Absent is the desired state.
|
|
216
|
+
}
|
|
147
217
|
fs.copyFileSync(staged, tmp);
|
|
148
218
|
fs.renameSync(tmp, canonicalFile(session.key));
|
|
149
219
|
fs.writeFileSync(join(dir, PI_VERSION_FILE), String(piVersion ?? ""));
|
|
220
|
+
// The two sidecars, immediately after the swap and under the same lock. NOT part of the swap
|
|
221
|
+
// itself, which is one rename and cannot be widened: what the lock buys them is that no
|
|
222
|
+
// other job can interleave, and what the ordering buys them is that they never describe a
|
|
223
|
+
// transcript older than the one now in place.
|
|
224
|
+
//
|
|
225
|
+
// EACH IS CAUGHT SEPARATELY, and that is not defensiveness for its own sake. These writes
|
|
226
|
+
// run AFTER the transcript is already promoted, so letting one throw would return
|
|
227
|
+
// `promote-failed` for a promotion that demonstrably happened -- a record that says the next
|
|
228
|
+
// run will cold start when it will in fact resume, which is worse than the bookkeeping loss
|
|
229
|
+
// it is reporting.
|
|
230
|
+
writeSidecar(dir, RESUME_CHAIN_FILE, session.key, chainValue(session));
|
|
231
|
+
writeContextSidecar(dir, session, context);
|
|
150
232
|
} finally {
|
|
151
233
|
fs.closeSync(fd);
|
|
152
234
|
try {
|
|
@@ -164,6 +246,79 @@ export function makeSessionStore({
|
|
|
164
246
|
}
|
|
165
247
|
}
|
|
166
248
|
|
|
249
|
+
/**
|
|
250
|
+
* The counter's next value. `session.resume` is the HOST's own decision to hand this key's transcript
|
|
251
|
+
* to a container, which is the only half of the exchange the container cannot influence; `resumed` (the
|
|
252
|
+
* container's verdict) is deliberately ignored for the counter and kept in the signature only because
|
|
253
|
+
* the record's own merge still wants it. A cold start resets, so a lineage always gets a fresh start
|
|
254
|
+
* from its next COMPLETED run -- a run that never completes promotes nothing and resets nothing, which
|
|
255
|
+
* is the safe direction: the key simply keeps cold-starting.
|
|
256
|
+
*/
|
|
257
|
+
function chainValue(session) {
|
|
258
|
+
return String(session.resume === true ? readResumeChain(session.key) + 1 : 0);
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
/**
|
|
262
|
+
* One sidecar write. Two properties, both deliberate.
|
|
263
|
+
*
|
|
264
|
+
* **It cannot write THROUGH a link.** `writeFileSync` follows one, which would turn a planted symlink
|
|
265
|
+
* in a key directory into a truncating write of any worker-writable file, with the container's own
|
|
266
|
+
* integers as the payload. Writing a temp and renaming over the name replaces whatever is there --
|
|
267
|
+
* link included -- with a regular file, and never opens the link's target. The temp is unlinked first
|
|
268
|
+
* for the same reason, since a planted link at THAT name would be the same hole one step along. The
|
|
269
|
+
* read side's `lstat` guard is the other half of this; neither is sufficient alone.
|
|
270
|
+
*
|
|
271
|
+
* **It is never fatal.** This runs AFTER the transcript is already promoted, so throwing would return
|
|
272
|
+
* `promote-failed` for a promotion that demonstrably happened, telling an operator the next run will
|
|
273
|
+
* cold start when it will in fact resume. The bookkeeping loss is logged and the truth is kept.
|
|
274
|
+
*/
|
|
275
|
+
function writeSidecar(dir, name, key, value) {
|
|
276
|
+
const file = join(dir, name);
|
|
277
|
+
const tmp = `${file}.incoming`;
|
|
278
|
+
try {
|
|
279
|
+
try {
|
|
280
|
+
fs.unlinkSync(tmp);
|
|
281
|
+
} catch {
|
|
282
|
+
// Absent is the desired state.
|
|
283
|
+
}
|
|
284
|
+
fs.writeFileSync(tmp, value);
|
|
285
|
+
fs.renameSync(tmp, file);
|
|
286
|
+
} catch (err) {
|
|
287
|
+
log("session_sidecar_failed", { key, file: name, reason: err?.message });
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* The context sidecar, whose three cases are all different.
|
|
293
|
+
*
|
|
294
|
+
* A run that RESUMED and measured nothing keeps the previous reading: the transcript it promoted is
|
|
295
|
+
* the old one extended, so the last real measurement is the closest true statement available, and a
|
|
296
|
+
* zero would read as "the context emptied", which cannot have happened.
|
|
297
|
+
*
|
|
298
|
+
* A COLD START, though, promoted a transcript that shares nothing with the one the old reading
|
|
299
|
+
* described, so the reading must GO. Keeping it is what turned a single high measurement into a key
|
|
300
|
+
* that refused itself forever: the gate read a stale number, cold-started, and the cold start left the
|
|
301
|
+
* same number behind for the next run to read. That loop had no exit that did not involve deleting the
|
|
302
|
+
* store by hand.
|
|
303
|
+
*/
|
|
304
|
+
function writeContextSidecar(dir, session, context) {
|
|
305
|
+
const file = join(dir, CONTEXT_FILE);
|
|
306
|
+
if (session.resume !== true) {
|
|
307
|
+
try {
|
|
308
|
+
fs.unlinkSync(file);
|
|
309
|
+
} catch {
|
|
310
|
+
// Absent is the desired state, so failing to remove what is not there is success.
|
|
311
|
+
}
|
|
312
|
+
return;
|
|
313
|
+
}
|
|
314
|
+
if (!context) return;
|
|
315
|
+
// The model rides along because the ratio is meaningless without it: a key is (kind, repo, ref) and
|
|
316
|
+
// carries no model, so two triggers on one issue can run different ones, and 25k tokens is 78% of a
|
|
317
|
+
// 32k window and 2.5% of a 1M one. A reading from another model is not a reading about this one.
|
|
318
|
+
const stamp = session.modelId ? ` ${session.modelId}` : "";
|
|
319
|
+
writeSidecar(dir, CONTEXT_FILE, session.key, `${context.tokens} ${context.window}${stamp}`);
|
|
320
|
+
}
|
|
321
|
+
|
|
167
322
|
function keyDir(key) {
|
|
168
323
|
return join(sessionsDir, key);
|
|
169
324
|
}
|
|
@@ -172,7 +327,7 @@ export function makeSessionStore({
|
|
|
172
327
|
}
|
|
173
328
|
|
|
174
329
|
/** The read path, gate by gate. The FIRST miss wins and names itself. */
|
|
175
|
-
function readCanonical(key, piVersion) {
|
|
330
|
+
function readCanonical(key, piVersion, modelId) {
|
|
176
331
|
const file = canonicalFile(key);
|
|
177
332
|
const check = inspectFile(file);
|
|
178
333
|
if (!check.ok) return COLD(check.reason);
|
|
@@ -184,26 +339,150 @@ export function makeSessionStore({
|
|
|
184
339
|
// repair that mid-run, so a version change is a cold start rather than a mid-run failure. An
|
|
185
340
|
// image that declares no version never resumes -- the safe direction, never "assume it matches".
|
|
186
341
|
if (piVersion === null) return COLD("pi-version-changed");
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
342
|
+
// Through the same guarded read as the two sidecars below it. This one predates them and was the
|
|
343
|
+
// one unguarded read left in the key directory; a symlink here would have decided a gate on the
|
|
344
|
+
// contents of some other file entirely.
|
|
345
|
+
const stamped = readSidecar(key, PI_VERSION_FILE);
|
|
346
|
+
if (stamped === null || stamped !== piVersion) return COLD("pi-version-changed");
|
|
347
|
+
|
|
348
|
+
// How many times in a row this key has already been resumed. Placed HERE, ahead of the header read,
|
|
349
|
+
// for two reasons. It is a small sidecar read exactly like the pi-version arm above it, so refusing
|
|
350
|
+
// on it skips pulling a transcript that may be megabytes; and unlike every other arm it asks about
|
|
351
|
+
// the LINEAGE rather than the file, so it needs nothing the file could tell it.
|
|
352
|
+
//
|
|
353
|
+
// The cost of that placement, stated rather than left to be discovered: a transcript that is both
|
|
354
|
+
// chain-exhausted AND corrupt reports the chain. That is the intentional refusal of the two, and the
|
|
355
|
+
// corruption is not hidden, only deferred -- this cold start's own promotion resets the counter, so
|
|
356
|
+
// the very next run reads the file and reports `unparseable`.
|
|
357
|
+
//
|
|
358
|
+
// FAILS OPEN on absence, which is the opposite of the age gate one arm down and deliberate. Every
|
|
359
|
+
// key that existed before this counter did has no file, and reading that as "already exhausted"
|
|
360
|
+
// would cold-start an operator's entire store the day they set the bound.
|
|
361
|
+
if (maxResumeChain > 0 && readResumeChain(key) >= maxResumeChain) return COLD("resume-chain-too-long");
|
|
362
|
+
|
|
363
|
+
// How full the context already is, against a ceiling the HOST owns. Not a duplicate of pi's own
|
|
364
|
+
// compaction threshold and deliberately not read from it: pi's is settable in a serviced repo's
|
|
365
|
+
// .pi/settings.json, so it is a line the repository can move, and this one cannot be. Past that
|
|
366
|
+
// threshold what a resumed job replays is not the transcript but a model-written summary of it,
|
|
367
|
+
// produced while that model was reading attacker-authored text (OQ-003), so this is a safety bound
|
|
368
|
+
// before it is an economic one.
|
|
369
|
+
//
|
|
370
|
+
// FAILS OPEN and INVENTS NO DENOMINATOR. No sidecar (every key promoted before this shipped, and
|
|
371
|
+
// every key under an image whose runner predates it), a compaction that left pi's own count
|
|
372
|
+
// unknown, or a window of zero all mean the gate has nothing to act on, and a gate with nothing to
|
|
373
|
+
// act on passes. A bytes-against-window guess was rejected rather than used as a fallback: the
|
|
374
|
+
// transcript is the whole branch INCLUDING what compaction folded away, so it over-reads exactly
|
|
375
|
+
// past the threshold this exists to catch, and there is no bytes-to-tokens calibration here to
|
|
376
|
+
// make it mean anything.
|
|
377
|
+
if (maxContextPct !== null) {
|
|
378
|
+
const seen = readContext(key);
|
|
379
|
+
// A reading STAMPED WITH ANOTHER MODEL is not a reading about this one, and using it is wrong in
|
|
380
|
+
// both directions: it refuses a job whose window is far larger than the one that was measured,
|
|
381
|
+
// and it passes one whose window is far smaller. Unknown on either side stays usable, so a
|
|
382
|
+
// deployment that names no model per trigger keeps the bound it had.
|
|
383
|
+
const foreign = seen !== null && seen.modelId !== null && modelId !== null && seen.modelId !== modelId;
|
|
384
|
+
if (seen !== null && !foreign && (seen.tokens * 100) / seen.window >= maxContextPct) return COLD("context-too-full");
|
|
192
385
|
}
|
|
193
|
-
if (stamped !== piVersion) return COLD("pi-version-changed");
|
|
194
386
|
|
|
195
|
-
// Cheapest real shape check, and the last one
|
|
196
|
-
// else the runner would throw on, so refusing here keeps
|
|
197
|
-
// surprises rather than for a file we could already tell
|
|
387
|
+
// Cheapest real shape check, and the last one before the header's own contents are used: the first
|
|
388
|
+
// line must be a pi session header. Anything else the runner would throw on, so refusing here keeps
|
|
389
|
+
// the container's degrade path for genuine surprises rather than for a file we could already tell
|
|
390
|
+
// was wrong.
|
|
391
|
+
let header = null;
|
|
198
392
|
try {
|
|
199
393
|
const head = String(fs.readFileSync(file, "utf8")).split("\n", 1)[0];
|
|
200
|
-
|
|
394
|
+
header = JSON.parse(head);
|
|
395
|
+
if (header?.type !== "session") return COLD("unparseable");
|
|
201
396
|
} catch {
|
|
202
397
|
return COLD("unparseable");
|
|
203
398
|
}
|
|
399
|
+
|
|
400
|
+
// The CONVERSATION's age, and it is a DIFFERENT CLOCK from `expired` above rather than a finer
|
|
401
|
+
// setting of it. The TTL reads the transcript's mtime, which the PROMOTE rename refreshes -- and only
|
|
402
|
+
// that: `copyFileSync` stamps its destination, never its source, so the resolve half leaves the
|
|
403
|
+
// canonical file's mtime alone (measured, because the obvious reading of the two call sites says
|
|
404
|
+
// otherwise). So `expired` is time since the last COMPLETED run on this key, and a lineage whose runs
|
|
405
|
+
// keep completing never expires however old its first turn is. pi's header carries the instant the
|
|
406
|
+
// session was created, so this
|
|
407
|
+
// costs no new persisted state -- the line is already read and parsed one gate up, and until now
|
|
408
|
+
// only its `type` was looked at.
|
|
409
|
+
//
|
|
410
|
+
// The arm is LAST because the earlier gates are cheaper and because a corrupt file is corrupt rather
|
|
411
|
+
// than old: `unparseable` must keep winning over this, or a damaged transcript would be reported as
|
|
412
|
+
// a lineage that aged out.
|
|
413
|
+
//
|
|
414
|
+
// UNREADABLE FAILS CLOSED, on the pi-version gate's precedent one arm up: a header with no usable
|
|
415
|
+
// timestamp cannot be shown to be young enough, and "assume it matches" is the direction that
|
|
416
|
+
// silently keeps resuming. Like `pi-version-changed`, one token covers all three causes (absent,
|
|
417
|
+
// wrong type, unparseable).
|
|
418
|
+
//
|
|
419
|
+
// A timestamp in the FUTURE passes, deliberately. It buys nothing to refuse one: the agent owns
|
|
420
|
+
// /session, so anything able to write a future timestamp is equally able to write the current one,
|
|
421
|
+
// and refusing would convert ordinary clock skew between a container and its host into a cold start
|
|
422
|
+
// for every key on the deployment.
|
|
423
|
+
if (maxAgeDays > 0) {
|
|
424
|
+
const started = Date.parse(typeof header.timestamp === "string" ? header.timestamp : "");
|
|
425
|
+
if (!Number.isFinite(started)) return COLD("conversation-too-old");
|
|
426
|
+
if (now() - started > maxAgeDays * 86400000) return COLD("conversation-too-old");
|
|
427
|
+
}
|
|
204
428
|
return { resume: true, reason: "resumed", bytes: check.bytes };
|
|
205
429
|
}
|
|
206
430
|
|
|
431
|
+
/**
|
|
432
|
+
* Every sidecar read goes through here, and it is the same load-bearing check `inspectFile` makes on
|
|
433
|
+
* the transcript: **`lstat`, regular files only.** The canonical store is host-only and never mounted,
|
|
434
|
+
* so nothing in a container can plant a link here -- but the directory NAME is derived rather than
|
|
435
|
+
* random (`sha256(kind, repo, ref)`), so anyone who knows the repository and the branch can compute it
|
|
436
|
+
* and pre-create the path. `readFileSync` and `writeFileSync` both follow links, which would turn a
|
|
437
|
+
* planted symlink into a read of any worker-readable file on the gate's path, and a promotion into a
|
|
438
|
+
* truncating write of any worker-writable one. The transcript has been guarded against exactly this
|
|
439
|
+
* since the feature shipped; these files inherit it rather than being the exception.
|
|
440
|
+
*
|
|
441
|
+
* SIZE-BOUNDED for the same reason the transcript is. Both formats are a handful of bytes, `maxBytes`
|
|
442
|
+
* does not cover them, and reading a 2.5 GiB file on the job's own path costs half a minute of wall
|
|
443
|
+
* clock before any container starts.
|
|
444
|
+
*/
|
|
445
|
+
function readSidecar(key, name) {
|
|
446
|
+
try {
|
|
447
|
+
const file = join(keyDir(key), name);
|
|
448
|
+
const st = fs.lstatSync(file);
|
|
449
|
+
if (!st.isFile() || st.size === 0 || st.size > SIDECAR_MAX_BYTES) return null;
|
|
450
|
+
return String(fs.readFileSync(file, "utf8")).trim();
|
|
451
|
+
} catch {
|
|
452
|
+
return null;
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
/**
|
|
457
|
+
* The consecutive-delivery counter for a key, or 0 when there is not a readable one. Never throws and
|
|
458
|
+
* never guesses: a missing, empty, corrupt or negative counter is 0, so the only way to be refused by
|
|
459
|
+
* the chain bound is for this store to have written a number that reaches it.
|
|
460
|
+
*/
|
|
461
|
+
function readResumeChain(key) {
|
|
462
|
+
const raw = readSidecar(key, RESUME_CHAIN_FILE);
|
|
463
|
+
if (raw === null) return 0;
|
|
464
|
+
const n = Number.parseInt(raw, 10);
|
|
465
|
+
// `String(n) === raw` is the same anti-truncation guard config.mjs applies to every integer knob,
|
|
466
|
+
// and it is what keeps a corrupt "3.5" from being read as a chain of three.
|
|
467
|
+
return Number.isInteger(n) && n > 0 && String(n) === raw ? n : 0;
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
/**
|
|
471
|
+
* The stored context occupancy for a key, or `null` when there is no measurement. Never throws, never
|
|
472
|
+
* guesses, and never returns a partial: anything it cannot read as two positive integers is no
|
|
473
|
+
* measurement at all, which the caller treats as "pass" rather than as zero.
|
|
474
|
+
*/
|
|
475
|
+
function readContext(key) {
|
|
476
|
+
const raw = readSidecar(key, CONTEXT_FILE);
|
|
477
|
+
if (raw === null) return null;
|
|
478
|
+
const [rawTokens, rawWindow, rawModel] = raw.split(/\s+/);
|
|
479
|
+
const tokens = Number.parseInt(rawTokens, 10);
|
|
480
|
+
const window = Number.parseInt(rawWindow, 10);
|
|
481
|
+
if (!Number.isInteger(tokens) || !Number.isInteger(window) || tokens < 0 || window <= 0) return null;
|
|
482
|
+
if (String(tokens) !== rawTokens || String(window) !== rawWindow) return null;
|
|
483
|
+
return { tokens, window, modelId: rawModel ?? null };
|
|
484
|
+
}
|
|
485
|
+
|
|
207
486
|
/**
|
|
208
487
|
* lstat, REGULAR FILES ONLY -- and this is the one line in the file that is load-bearing security
|
|
209
488
|
* rather than hygiene.
|
package/src/start.mjs
CHANGED
|
@@ -13,6 +13,7 @@ import { makeForgejoAuth } from "./forgejo-auth.mjs";
|
|
|
13
13
|
import { makeForgejoHost } from "./forgejo-host.mjs";
|
|
14
14
|
import { makeAzureAuth } from "./azure-auth.mjs";
|
|
15
15
|
import { makeAzureHost } from "./azure-host.mjs";
|
|
16
|
+
import { makeEgressPreflight } from "./egress.mjs";
|
|
16
17
|
import { makeImagePreflight } from "./image-preflight.mjs";
|
|
17
18
|
import { createWorker } from "./index.mjs";
|
|
18
19
|
import { makeCollectChain } from "./outbox.mjs";
|
|
@@ -112,6 +113,22 @@ export function makeReaper({ log }) {
|
|
|
112
113
|
await exec("docker", ["rm", "-f", name]);
|
|
113
114
|
log("reaped_container", { name });
|
|
114
115
|
}
|
|
116
|
+
// REQ-EGRESS-ALLOWLIST: the per-job networks those containers were on. Swept AFTER the containers,
|
|
117
|
+
// because a network with a member still attached cannot be removed -- and swept by the SAME
|
|
118
|
+
// `pi-job-` filter, so the namespace rule that keeps an operator's live sandbox safe from the
|
|
119
|
+
// container reaper keeps their sandbox NETWORK safe too, with no second rule to remember.
|
|
120
|
+
//
|
|
121
|
+
// A crashed worker is the case this exists for: `runContainer`'s own finally removes the network
|
|
122
|
+
// on every ordinary path, so anything still here outlived a process that did not get to run it.
|
|
123
|
+
// A network still in use by something else fails to remove and is skipped, which is correct: this
|
|
124
|
+
// is a best-effort sweep and never a reason not to boot.
|
|
125
|
+
const { stdout: nets } = await exec("docker", ["network", "ls", "--filter", "name=pi-job-", "--format", "{{.Name}}"]);
|
|
126
|
+
for (const net of nets.split("\n").map((n) => n.trim()).filter(Boolean)) {
|
|
127
|
+
try {
|
|
128
|
+
await exec("docker", ["network", "rm", net]);
|
|
129
|
+
log("reaped_network", { network: net });
|
|
130
|
+
} catch {} // still in use, or already gone -- either way not this boot's problem
|
|
131
|
+
}
|
|
115
132
|
} catch (err) {
|
|
116
133
|
log("reaper_skipped", { reason: err?.message });
|
|
117
134
|
}
|
|
@@ -146,6 +163,7 @@ export async function startWorker(
|
|
|
146
163
|
makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
|
|
147
164
|
makeRunContainer: makeRunContainerFn = makeRunContainer,
|
|
148
165
|
makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
|
|
166
|
+
makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
|
|
149
167
|
makeGitLabAuth: makeGitLabAuthFn = makeGitLabAuth,
|
|
150
168
|
makeGitLabHost: makeGitLabHostFn = makeGitLabHost,
|
|
151
169
|
makeForgejoAuth: makeForgejoAuthFn = makeForgejoAuth,
|
|
@@ -280,6 +298,9 @@ export async function startWorker(
|
|
|
280
298
|
sessionsDir: config.sessionsDir,
|
|
281
299
|
ttlDays: config.sessionsTtlDays,
|
|
282
300
|
maxBytes: config.sessionMaxBytes,
|
|
301
|
+
maxAgeDays: config.sessionMaxAgeDays,
|
|
302
|
+
maxResumeChain: config.sessionMaxResumeChain,
|
|
303
|
+
maxContextPct: config.sessionMaxContextPct,
|
|
283
304
|
log,
|
|
284
305
|
});
|
|
285
306
|
// Boot sweep, beside the log reaper and for the same reason it is beside rather than inside it: these
|
|
@@ -386,6 +407,12 @@ export async function startWorker(
|
|
|
386
407
|
// one who removes it would stay admitted. Contrast the staged-package manifest, correctly read once at
|
|
387
408
|
// boot because it is deploy-time state under a :ro mount; the host's image set is not.
|
|
388
409
|
imagePreflight: makeImagePreflightFn({ image: config.jobImage }),
|
|
410
|
+
// REQ-EGRESS-ALLOWLIST, and built here for the same reason the image preflight is: one deployment
|
|
411
|
+
// value, one place, so the gate that checks the proxy and the runner that attaches to its network
|
|
412
|
+
// cannot disagree about which proxy is meant. Nothing is memoised here either -- an operator who
|
|
413
|
+
// starts the proxy mid-day must not stay refused, and one who stops it must not stay admitted.
|
|
414
|
+
// Unarmed it spawns nothing at all, so a deployment without a policy pays for none of this.
|
|
415
|
+
egressPreflight: makeEgressPreflightFn({ proxy: config.egressProxy, armed: config.egress }),
|
|
389
416
|
// Completed-only, so a policy or infra exit leaves the canonical transcript byte-identical and a
|
|
390
417
|
// retry starts from what the first attempt did (CONST-RETRY-INFRA-ONLY).
|
|
391
418
|
promoteSession: sessionStore.promoteSession,
|
|
@@ -397,6 +424,8 @@ export async function startWorker(
|
|
|
397
424
|
runContainer: makeRunContainerFn({
|
|
398
425
|
image: config.jobImage,
|
|
399
426
|
hostEnv: env,
|
|
427
|
+
egress: config.egress, // REQ-EGRESS-ALLOWLIST: the per-job network and the proxy variables
|
|
428
|
+
egressProxy: config.egressProxy,
|
|
400
429
|
openJobLog,
|
|
401
430
|
globalPiDir: config.globalPiDir, // REQ-GLOBAL-PI-OVERLAY: :ro overlay mount when configured
|
|
402
431
|
allowGlobalExtensions: config.allowGlobalExtensions,
|
package/src/triggers.mjs
CHANGED
|
@@ -278,8 +278,9 @@ function validatePackagesFlag(run, at, path) {
|
|
|
278
278
|
* had. `validateReplicas` states the argument in one line and it applies verbatim here: a field accepted
|
|
279
279
|
* where it does nothing is how an operator comes to trust one that does nothing.
|
|
280
280
|
*
|
|
281
|
-
* "NOT YET COVERED", not impossible --
|
|
282
|
-
*
|
|
281
|
+
* "NOT YET COVERED", not impossible -- a distinction this file keeps because the two are different facts
|
|
282
|
+
* and an operator planning work needs the right one. `run.replicas` carried the same wording for the same
|
|
283
|
+
* reason until #187 closed its gap; this one is still open, which is why the phrasing outlived it. The local key already exists and is
|
|
283
284
|
* the strongest key in this feature: session-key.mjs keys a cron job on its scheduler id, which is
|
|
284
285
|
* operator-authored, unique across the file, stable across fires, and chosen by nobody untrusted. Nothing
|
|
285
286
|
* reaches it. Wiring `resolveSession` into the local path is a feature, and this line is what stops the
|
|
@@ -427,8 +428,9 @@ const INSTRUCTIONS_MAX = 2000;
|
|
|
427
428
|
* `run.instructions` (issue #60): one line of operator standing text, rendered into the USER prompt's
|
|
428
429
|
* envelope above the fenced data region.
|
|
429
430
|
*
|
|
430
|
-
* REFUSED on cron, and it is a DIFFERENT refusal from
|
|
431
|
-
* an operator-authored free-text field landing in the same
|
|
431
|
+
* REFUSED on cron, and it is a DIFFERENT refusal from the "not yet covered" shape (run.resume's, since
|
|
432
|
+
* #187 retired run.replicas'): cron already has an operator-authored free-text field landing in the same
|
|
433
|
+
* region of the same file. A local job's prompt
|
|
432
434
|
* is `flow hint + pointer + run.task` with no envelope, no data heading and no fence (prepare.mjs), so
|
|
433
435
|
* there is no "standing" region distinct from the task for a second field to occupy. Two fields writing
|
|
434
436
|
* one region with an undefined combination order is worse than a field that does nothing, because both
|
|
@@ -597,12 +599,12 @@ function validateRepository(run, onType, at, path) {
|
|
|
597
599
|
*
|
|
598
600
|
* WHY EACH REFUSAL:
|
|
599
601
|
* - a LOCAL (cron) trigger: its `/workspace` IS the operator's folder, bind-mounted read-write and edited
|
|
600
|
-
* in place, so two replicas would stomp each other's working tree with no gate and no undo. A
|
|
602
|
+
* in place, so two replicas would stomp each other's working tree with no gate and no undo. A forge
|
|
601
603
|
* job gets its own `mkdtemp`'d clone, which is the entire reason this is safe there and not here.
|
|
602
|
-
* Checked FIRST
|
|
603
|
-
*
|
|
604
|
-
*
|
|
605
|
-
*
|
|
604
|
+
* Checked FIRST, and since #187 that ordering carries the whole kind gate rather than merely picking
|
|
605
|
+
* which reason a cron trigger hears. Every forge mints its branch through the same `issueBranch`, so
|
|
606
|
+
* anything that is not `local` is now allowed; move this below the range check and a cron entry
|
|
607
|
+
* carrying `replicas: 2` would be ACCEPTED, not refused with a different message.
|
|
606
608
|
* - a non-integer, `< 2`, or `> REPLICAS_MAX`. `1` is REFUSED rather than accepted: a one-member replica
|
|
607
609
|
* set is a field that does nothing, and this validator's whole job is to make sure nothing does nothing.
|
|
608
610
|
* - `run.resume: true`. A resumed run continues ONE lineage; replicas exist to fork it. This is the
|
|
@@ -624,9 +626,6 @@ function validateReplicas(run, at, path) {
|
|
|
624
626
|
if (run.kind === "local") {
|
|
625
627
|
throw configError(`${at}: run.replicas is not available on a cron trigger -- a local job's /workspace IS the operator's folder, bind-mounted read-write, so two replicas would edit one working tree with no gate and no undo: ${path}`);
|
|
626
628
|
}
|
|
627
|
-
if (run.kind !== "github") {
|
|
628
|
-
throw configError(`${at}: run.replicas is not yet covered for ${run.kind} triggers (github only in this version); every forge mints its branch the same way, so this is a gap to close, not a limit: ${path}`);
|
|
629
|
-
}
|
|
630
629
|
if (!Number.isInteger(replicas) || replicas < 2 || replicas > REPLICAS_MAX) {
|
|
631
630
|
throw configError(`${at}: run.replicas must be an integer between 2 and ${REPLICAS_MAX} when present -- ${REPLICAS_MAX} is the ceiling because PI_CONCURRENCY defaults to 3, so a further replica would queue instead of racing, and 1 is refused because a one-member replica set is a flag that does nothing (got ${JSON.stringify(replicas)}): ${path}`);
|
|
632
631
|
}
|
package/src/up.mjs
CHANGED
|
@@ -24,6 +24,7 @@ import { spawn as nodeSpawn } from "node:child_process";
|
|
|
24
24
|
import { chmodSync, existsSync, readFileSync, renameSync, statSync, writeFileSync } from "node:fs";
|
|
25
25
|
import { connect as netConnect } from "node:net";
|
|
26
26
|
import { join } from "node:path";
|
|
27
|
+
import { egressArmed as egressArmedFn } from "./egress.mjs";
|
|
27
28
|
import { updateEnvFile } from "./env-file.mjs";
|
|
28
29
|
|
|
29
30
|
// The one image up may ever fetch, and the local name jobs run under. Literal on purpose (not
|
|
@@ -63,6 +64,31 @@ const VALKEY_RUN_ARGS = [
|
|
|
63
64
|
"yes",
|
|
64
65
|
];
|
|
65
66
|
|
|
67
|
+
// deploy/docker-compose.yml's `egress` profile, reproduced as one docker run (REQ-EGRESS-ALLOWLIST):
|
|
68
|
+
// same digest-pinned image, same two mounts, same explicit container name, same restart policy, on the
|
|
69
|
+
// same upstream network. Written out here for the same reason VALKEY_RUN_ARGS is -- an operator who runs
|
|
70
|
+
// `up` and one who runs compose must end up with the same component, and two ways of starting one thing
|
|
71
|
+
// is two places for it to drift.
|
|
72
|
+
//
|
|
73
|
+
// The per-job networks are NOT here: the worker creates one per job and attaches this container to it for
|
|
74
|
+
// the life of that run. This is only the proxy and its way out.
|
|
75
|
+
const EGRESS_NETWORK_ARGS = ["network", "create", "pi-dispatch-egress-out"];
|
|
76
|
+
const EGRESS_RUN_ARGS = [
|
|
77
|
+
"run",
|
|
78
|
+
"-d",
|
|
79
|
+
"--name",
|
|
80
|
+
"pi-dispatch-egress-proxy",
|
|
81
|
+
"--restart",
|
|
82
|
+
"unless-stopped",
|
|
83
|
+
"--network",
|
|
84
|
+
"pi-dispatch-egress-out",
|
|
85
|
+
"-v",
|
|
86
|
+
"./deploy/egress-proxy.conf:/etc/squid/squid.conf:ro",
|
|
87
|
+
"-v",
|
|
88
|
+
"./egress-allowlist.conf:/etc/pi-dispatch/allowlist.conf:ro",
|
|
89
|
+
"ubuntu/squid@sha256:6a097f68bae708cedbabd6188d68c7e2e7a38cedd05a176e1cc0ba29e3bbe029",
|
|
90
|
+
];
|
|
91
|
+
|
|
66
92
|
export async function runUp(argv = [], deps = {}) {
|
|
67
93
|
const {
|
|
68
94
|
env = process.env,
|
|
@@ -173,6 +199,38 @@ export async function runUp(argv = [], deps = {}) {
|
|
|
173
199
|
summary.push(["WEBHOOK_SECRET", "no .env here — skipped (set it wherever your env lives)"]);
|
|
174
200
|
}
|
|
175
201
|
|
|
202
|
+
// (e2) the egress policy's proxy, and ONLY when the operator has already armed it. up never invents
|
|
203
|
+
// operator policy -- the same doctrine that keeps it pulling this repo's own image and no other -- so a
|
|
204
|
+
// deployment that has not set PI_EGRESS hears nothing about this at all.
|
|
205
|
+
//
|
|
206
|
+
// AFTER init, deliberately: init has just scaffolded egress-allowlist.conf, and starting a proxy whose
|
|
207
|
+
// allowlist file does not exist gets a directory created by docker where a file belonged and a squid
|
|
208
|
+
// that fails confusingly. If the file is still missing, this step declines itself and says which file.
|
|
209
|
+
if (egressArmedFn(env)) {
|
|
210
|
+
if ((await runCmd(spawn, "docker", ["inspect", "--format={{.State.Running}}", "pi-dispatch-egress-proxy"])) === 0) {
|
|
211
|
+
out("\n✓ Egress proxy already present (pi-dispatch-egress-proxy)\n");
|
|
212
|
+
summary.push(["egress", "proxy already present — left untouched"]);
|
|
213
|
+
} else if (!fs.existsSync(join(cwd, "egress-allowlist.conf"))) {
|
|
214
|
+
out("\n✗ the egress policy is on but egress-allowlist.conf is not here — not starting a proxy with no allowlist\n");
|
|
215
|
+
summary.push(["egress", "skipped — no egress-allowlist.conf in this folder; run `pi-dispatch init` here, then `up` again"]);
|
|
216
|
+
} else if (
|
|
217
|
+
await consent("The egress policy is on (PI_EGRESS=0 opts out) but the allowlist proxy is not running. up would start it (same semantics as deploy/docker-compose.yml --profile egress):", [`docker ${EGRESS_NETWORK_ARGS.join(" ")}`, `docker ${quoteArgs(EGRESS_RUN_ARGS)}`], { yes, out, prompt })
|
|
218
|
+
) {
|
|
219
|
+
// The network may already exist from a previous run; that is not a failure, so its code is not
|
|
220
|
+
// checked. The proxy is what matters and it is checked.
|
|
221
|
+
await runStreamed(spawn, "docker", EGRESS_NETWORK_ARGS, out);
|
|
222
|
+
if ((await runStreamed(spawn, "docker", EGRESS_RUN_ARGS, out)) !== 0) {
|
|
223
|
+
out("✗ could not start the egress proxy — continuing; doctor below will re-check it\n");
|
|
224
|
+
summary.push(["egress", "start failed — every job refuses pre-spend until it is up (costs no budget, runs nothing)"]);
|
|
225
|
+
} else {
|
|
226
|
+
summary.push(["egress", "started pi-dispatch-egress-proxy on pi-dispatch-egress-out"]);
|
|
227
|
+
}
|
|
228
|
+
} else {
|
|
229
|
+
out("skipped — start it later with `docker compose -f deploy/docker-compose.yml --profile egress up -d`\n");
|
|
230
|
+
summary.push(["egress", "skipped (declined) — every job is refused pre-spend until the proxy is up (PI_EGRESS=0 opts out)"]);
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
|
|
176
234
|
// (f) doctor — always, verbatim: up converges what it can, doctor is the judge of what remains
|
|
177
235
|
// (provider key, forge env, overlay …), and its verdict is up's exit code.
|
|
178
236
|
out("\ndoctor:\n");
|