@edgehero/pi-dispatch 3.0.0 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +34 -0
- package/package.json +6 -2
- package/src/backend-local.mjs +69 -0
- package/src/backend-podman.mjs +44 -13
- package/src/config.mjs +39 -1
- package/src/container-spec.mjs +70 -7
- package/src/cpu-reserve.mjs +344 -0
- package/src/daemon-facts.mjs +58 -0
- package/src/docker-run.mjs +83 -6
- package/src/doctor.mjs +631 -21
- package/src/env-allowlist.mjs +23 -5
- package/src/host-budget.mjs +736 -0
- package/src/host-pi.mjs +1 -1
- package/src/index.mjs +237 -65
- package/src/job-size.mjs +286 -0
- package/src/job-user.mjs +66 -5
- package/src/live-probes.mjs +150 -25
- package/src/model-catalog.mjs +1 -1
- package/src/model-endpoints.mjs +22 -0
- package/src/models-json.mjs +10 -4
- package/src/output-cap.mjs +3 -3
- package/src/prepare.mjs +7 -3
- package/src/processor.mjs +91 -14
- package/src/reserved-env.mjs +3 -2
- package/src/run-container.mjs +62 -8
- package/src/run-history.mjs +204 -117
- package/src/sandbox-store.mjs +5 -1
- package/src/sandbox.mjs +48 -7
- package/src/scoped-limits.mjs +94 -10
- package/src/size-records.mjs +80 -0
- package/src/size-suggest.mjs +441 -0
- package/src/start.mjs +133 -9
- package/src/triggers.mjs +8 -5
package/src/run-history.mjs
CHANGED
|
@@ -6,6 +6,7 @@ import { resolveBackendName } from "./backend-registry.mjs";
|
|
|
6
6
|
import { isForgeKind, targetSeparator } from "./forges.mjs";
|
|
7
7
|
import { MODEL_REF_PATTERN as USAGE_ID_PATTERN } from "./model-ref.mjs";
|
|
8
8
|
import { isProjectId } from "./project-id.mjs";
|
|
9
|
+
import { recordedJobSize } from "./job-size.mjs";
|
|
9
10
|
|
|
10
11
|
/**
|
|
11
12
|
* Durable per-run history.
|
|
@@ -13,7 +14,7 @@ import { isProjectId } from "./project-id.mjs";
|
|
|
13
14
|
* The PURE HELPERS are total functions over their arguments -- no filesystem, no clock, no
|
|
14
15
|
* `process.env`, no randomness -- so the record shape and the filename/telemetry parsing are testable
|
|
15
16
|
* without a container, a queue, or a disk: `sanitizeJobId`, `parseExitTurns`, `parseExitTokens`,
|
|
16
|
-
* `parseExitSession`, `parseExitUsage`, `buildRecord`.
|
|
17
|
+
* `parseExitSession`, `parseExitUsage`, `parseExitResources`, `buildRecord`.
|
|
17
18
|
*
|
|
18
19
|
* The I/O factories -- `makeLogSink`, `makeRecordWriter`, `makeFindPreviousRun`, `makeLogReaper` --
|
|
19
20
|
* each inject their own `fs` (and, for the reaper, their own clock via `now`), so they too are testable
|
|
@@ -64,7 +65,7 @@ export function sanitizeJobId(id) {
|
|
|
64
65
|
* END (a mid-write death) has no complete object and is skipped too. A fragment is never MISREAD as a
|
|
65
66
|
* value: a broken one does not parse and an anchorless one is not repaired.
|
|
66
67
|
*
|
|
67
|
-
* NEVER throws, like the
|
|
68
|
+
* NEVER throws, like `decisiveExitLine`, the one caller every scanner goes through.
|
|
68
69
|
*/
|
|
69
70
|
function parseTailLine(line) {
|
|
70
71
|
try {
|
|
@@ -83,6 +84,43 @@ function parseTailLine(line) {
|
|
|
83
84
|
return null;
|
|
84
85
|
}
|
|
85
86
|
|
|
87
|
+
/** The signed marker the image's supervisor puts on every exit line it writes (image/runner/src/supervise.mjs). */
|
|
88
|
+
export const EXIT_BY_SUPERVISOR = "supervisor";
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* THE exit line every `parseExit*` scanner reads (issue #596), parsed, or null when `text` holds none: "the decisive
|
|
92
|
+
* exit line" everywhere in this file and in the specs means this function's pick, never simply the last line. One
|
|
93
|
+
* picker, so the scanners cannot disagree about which line decided the run.
|
|
94
|
+
*
|
|
95
|
+
* Normally the LAST `exit` event, found from the end with `parseTailLine`'s glue repair. One exception: a line the
|
|
96
|
+
* image's supervisor wrote (`by: "supervisor"`) when an EARLIER exit line exists that is not the supervisor's. That
|
|
97
|
+
* order means the runner wrote its own line and was killed afterwards (a tool can SIGKILL it after its decided line, and
|
|
98
|
+
* so can the kernel), and the supervisor then reported the death. The runner's line is the run's real outcome, with the
|
|
99
|
+
* tokens, usage and session the supervisor never has, so it decides, and the supervisor's line is ignored by every
|
|
100
|
+
* scanner: the run reads exactly as it did before the supervisor existed (its container exit `137` still retries).
|
|
101
|
+
* Read the other way round, the supervisor's `oom-killed` would turn a finished run into a stopped one, its tokens lost
|
|
102
|
+
* from the daily cap and its dollars settled at the floor.
|
|
103
|
+
*
|
|
104
|
+
* With a key the text is `authenticExitLines`' output, so both lines are signed and `by` is inside the MAC. Without
|
|
105
|
+
* one nothing here is trusted anyway: a tool that writes a `by: "supervisor"` line after the runner's only makes the
|
|
106
|
+
* runner's line decide, which it could have arranged by writing nothing. NEVER throws.
|
|
107
|
+
*/
|
|
108
|
+
function decisiveExitLine(text) {
|
|
109
|
+
if (typeof text !== "string") return null;
|
|
110
|
+
const lines = text.split("\n");
|
|
111
|
+
let last = null;
|
|
112
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
113
|
+
const line = lines[i].trim();
|
|
114
|
+
if (line === "") continue;
|
|
115
|
+
const parsed = parseTailLine(line);
|
|
116
|
+
if (parsed?.event !== "exit") continue;
|
|
117
|
+
if (parsed?.by !== EXIT_BY_SUPERVISOR) return parsed;
|
|
118
|
+
// The supervisor's line: it decides only if nothing the runner wrote comes before it.
|
|
119
|
+
if (last === null) last = parsed;
|
|
120
|
+
}
|
|
121
|
+
return last;
|
|
122
|
+
}
|
|
123
|
+
|
|
86
124
|
/**
|
|
87
125
|
* Recover the agent's turn count from buffered container stdout, or `null` if it is not reported.
|
|
88
126
|
*
|
|
@@ -90,9 +128,9 @@ function parseTailLine(line) {
|
|
|
90
128
|
* own lines. Only the decided-outcome exit line carries `turns` (`image/runner/run-job.mjs:431`); the
|
|
91
129
|
* catch-path exit line (`:446`) omits it. The decided line's `retryTurns` (issue #449, pi's own
|
|
92
130
|
* auto-retry turns, which the turn budget does not count) is not recovered here: it is diagnostic, read
|
|
93
|
-
* from the container log, and the record's `turns` stays the budgeted count.
|
|
94
|
-
*
|
|
95
|
-
*
|
|
131
|
+
* from the container log, and the record's `turns` stays the budgeted count. Return the turns of the
|
|
132
|
+
* decisive exit line (`decisiveExitLine`, which repairs a glued line on the way through `parseTailLine`)
|
|
133
|
+
* when it reports an integer count.
|
|
96
134
|
*
|
|
97
135
|
* This is read-only telemetry: it MUST NEVER throw and MUST NOT feed exit-code or retry
|
|
98
136
|
* classification -- that is the container exit code's job (INT-RUNNER-EXIT-CODE-PROTOCOL). Every parse
|
|
@@ -101,16 +139,9 @@ function parseTailLine(line) {
|
|
|
101
139
|
* container exit code alone decides the class, and that parsed reason only picks a label INSIDE exit 2.
|
|
102
140
|
*/
|
|
103
141
|
export function parseExitTurns(text) {
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
const line = lines[i].trim();
|
|
108
|
-
if (line === "") continue;
|
|
109
|
-
const parsed = parseTailLine(line);
|
|
110
|
-
if (parsed?.event !== "exit") continue;
|
|
111
|
-
return Number.isInteger(parsed?.turns) ? parsed.turns : null;
|
|
112
|
-
}
|
|
113
|
-
return null;
|
|
142
|
+
const parsed = decisiveExitLine(text);
|
|
143
|
+
if (parsed === null) return null;
|
|
144
|
+
return Number.isInteger(parsed?.turns) ? parsed.turns : null;
|
|
114
145
|
}
|
|
115
146
|
|
|
116
147
|
/**
|
|
@@ -132,29 +163,21 @@ export function parseExitTurns(text) {
|
|
|
132
163
|
export const RUNNER_POLICY_REASONS = new Set(["provider-auth-refused", "cost-cap", "model-not-allowed", "cost-cap-unenforceable", "model-policy-unenforceable"]);
|
|
133
164
|
|
|
134
165
|
/**
|
|
135
|
-
* The reason off the
|
|
136
|
-
* that same line says `code: 2`.
|
|
137
|
-
* line through `parseTailLine`, and NEVER throws.
|
|
166
|
+
* The reason off the decisive exit line (`decisiveExitLine`), or `null`: a member of `RUNNER_POLICY_REASONS`
|
|
167
|
+
* and only when that same line says `code: 2`. Read off the same line as `parseExitTurns`, and NEVER throws.
|
|
138
168
|
*
|
|
139
169
|
* Why this may reach the outcome when its five siblings may not: it cannot change the retry class. The
|
|
140
170
|
* processor consults it only inside its container-exit-2 branch, so a container that exits 1 while
|
|
141
171
|
* printing `{"event":"exit","code":2,"reason":"provider-auth-refused"}` is still InfraRetry, and the
|
|
142
172
|
* worst a forged line can do on a real exit 2 is swap one not-retried label for another from a closed
|
|
143
173
|
* set. The `code === 2` check on the line itself is what keeps a stale or mismatched line (a runner
|
|
144
|
-
* that said 1 in its own words) from naming the outcome. Only the
|
|
145
|
-
*
|
|
174
|
+
* that said 1 in its own words) from naming the outcome. Only the decisive exit line counts, because any
|
|
175
|
+
* other in the tail is not the one the runner exited on. A fixed enum, so the PII-free record stays so.
|
|
146
176
|
*/
|
|
147
177
|
export function parseExitReason(text) {
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
const line = lines[i].trim();
|
|
152
|
-
if (line === "") continue;
|
|
153
|
-
const parsed = parseTailLine(line);
|
|
154
|
-
if (parsed?.event !== "exit") continue;
|
|
155
|
-
return parsed?.code === 2 && RUNNER_POLICY_REASONS.has(parsed?.reason) ? parsed.reason : null;
|
|
156
|
-
}
|
|
157
|
-
return null;
|
|
178
|
+
const parsed = decisiveExitLine(text);
|
|
179
|
+
if (parsed === null) return null;
|
|
180
|
+
return parsed?.code === 2 && RUNNER_POLICY_REASONS.has(parsed?.reason) ? parsed.reason : null;
|
|
158
181
|
}
|
|
159
182
|
|
|
160
183
|
/**
|
|
@@ -165,8 +188,8 @@ export function parseExitReason(text) {
|
|
|
165
188
|
export const COST_CAP_WHYS = Object.freeze(["unboundable", "external", "over-cap"]);
|
|
166
189
|
|
|
167
190
|
/**
|
|
168
|
-
* The `why` off the
|
|
169
|
-
* `code: 2` and `reason: "cost-cap"`.
|
|
191
|
+
* The `why` off the decisive exit line (`decisiveExitLine`), or `null`: a member of `COST_CAP_WHYS`, and only when that
|
|
192
|
+
* same line says `code: 2` and `reason: "cost-cap"`. Read off the same line as `parseExitReason`, and NEVER throws.
|
|
170
193
|
*
|
|
171
194
|
* The line is container-written, so nothing it carries is trusted as text: a value outside the closed set is null,
|
|
172
195
|
* never a string copied through. It feeds no classification. The processor reads it only in its exit-2 branch, and
|
|
@@ -174,46 +197,32 @@ export const COST_CAP_WHYS = Object.freeze(["unboundable", "external", "over-cap
|
|
|
174
197
|
* wrong rule of the three.
|
|
175
198
|
*/
|
|
176
199
|
export function parseExitWhy(text) {
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
const line = lines[i].trim();
|
|
181
|
-
if (line === "") continue;
|
|
182
|
-
const parsed = parseTailLine(line);
|
|
183
|
-
if (parsed?.event !== "exit") continue;
|
|
184
|
-
return parsed?.code === 2 && parsed?.reason === "cost-cap" && COST_CAP_WHYS.includes(parsed?.why) ? parsed.why : null;
|
|
185
|
-
}
|
|
186
|
-
return null;
|
|
200
|
+
const parsed = decisiveExitLine(text);
|
|
201
|
+
if (parsed === null) return null;
|
|
202
|
+
return parsed?.code === 2 && parsed?.reason === "cost-cap" && COST_CAP_WHYS.includes(parsed?.why) ? parsed.why : null;
|
|
187
203
|
}
|
|
188
204
|
|
|
189
205
|
/**
|
|
190
|
-
* The
|
|
191
|
-
* the
|
|
206
|
+
* The decisive exit line's own `code` (`decisiveExitLine`), an integer, or `null` (issue #501, PR #542's review round
|
|
207
|
+
* 3). Read off the same line as its siblings, and NEVER throws.
|
|
192
208
|
*
|
|
193
209
|
* Read for ONE purpose: the dollar settlement trusts an exit line's cost only when the line's `code` equals the
|
|
194
210
|
* container's real exit code (processor.mjs). The runner writes the code it exits with on both exit lines
|
|
195
211
|
* (`image/runner/run-job.mjs`: `...capExitMessage(outcome)` then `return outcome.code` on the decided path, and
|
|
196
|
-
* `code: capped.code` on the catch path; pinned by a test), so
|
|
212
|
+
* `code: capped.code` on the catch path; pinned by a test), so the runner's genuine line always matches, while a line a
|
|
197
213
|
* job's own tool forged before a `docker stop` (exit 137, no genuine line after it) does not. It never feeds the
|
|
198
214
|
* retry class (INT-RUNNER-EXIT-CODE-PROTOCOL).
|
|
199
215
|
*/
|
|
200
216
|
export function parseExitCode(text) {
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
const line = lines[i].trim();
|
|
205
|
-
if (line === "") continue;
|
|
206
|
-
const parsed = parseTailLine(line);
|
|
207
|
-
if (parsed?.event !== "exit") continue;
|
|
208
|
-
return Number.isSafeInteger(parsed?.code) ? parsed.code : null;
|
|
209
|
-
}
|
|
210
|
-
return null;
|
|
217
|
+
const parsed = decisiveExitLine(text);
|
|
218
|
+
if (parsed === null) return null;
|
|
219
|
+
return Number.isSafeInteger(parsed?.code) ? parsed.code : null;
|
|
211
220
|
}
|
|
212
221
|
|
|
213
222
|
/**
|
|
214
223
|
* Recover the agent's token usage from buffered container stdout, or `null` if it is not reported.
|
|
215
224
|
*
|
|
216
|
-
* Mirrors `parseExitTurns`:
|
|
225
|
+
* Mirrors `parseExitTurns`: read the `tokens` object of the decisive exit line (`decisiveExitLine`)
|
|
217
226
|
* (`{ input, output, total, cost }`). Only the success exit line carries it
|
|
218
227
|
* (`image/runner/run-job.mjs`); the catch-path exit line omits it, and a container that died before the
|
|
219
228
|
* runner's exit line yields none -- all three cases are `null`.
|
|
@@ -277,25 +286,18 @@ export const SESSION_REASONS = new Set([
|
|
|
277
286
|
* `SESSION_REASONS` check below, that sentence is enforced rather than merely intended.
|
|
278
287
|
*/
|
|
279
288
|
export function parseExitSession(text) {
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
// the comment above claimed "a boolean and a fixed enum". An unrecognised token reads as `null`
|
|
293
|
-
// (the runner said nothing this contract can represent) rather than being carried through: the
|
|
294
|
-
// enum is documented CLOSED in INT-RUN-HISTORY-FILE-CONTRACT, so a value outside it was already
|
|
295
|
-
// contract-violating and every consumer already handles null.
|
|
296
|
-
return { resumed: sess.resumed, reason: SESSION_REASONS.has(sess.reason) ? sess.reason : null };
|
|
297
|
-
}
|
|
298
|
-
return null;
|
|
289
|
+
const parsed = decisiveExitLine(text);
|
|
290
|
+
if (parsed === null) return null;
|
|
291
|
+
const sess = parsed?.session;
|
|
292
|
+
if (sess && typeof sess === "object" && !Array.isArray(sess) && typeof sess.resumed === "boolean") {
|
|
293
|
+
// The reason is checked against the CLOSED enum, not merely against `typeof === "string"`, which
|
|
294
|
+
// is what it used to be. The container owns this value, so an unchecked string put an
|
|
295
|
+
// attacker-shapeable one into a record whose PII-free property rests on holding none -- while
|
|
296
|
+
// the comment above claimed "a boolean and a fixed enum". An unrecognised token reads as `null`
|
|
297
|
+
// (the runner said nothing this contract can represent) rather than being carried through: the
|
|
298
|
+
// enum is documented CLOSED in INT-RUN-HISTORY-FILE-CONTRACT, so a value outside it was already
|
|
299
|
+
// contract-violating and every consumer already handles null.
|
|
300
|
+
return { resumed: sess.resumed, reason: SESSION_REASONS.has(sess.reason) ? sess.reason : null };
|
|
299
301
|
}
|
|
300
302
|
return null;
|
|
301
303
|
}
|
|
@@ -311,27 +313,86 @@ export function parseExitSession(text) {
|
|
|
311
313
|
* reads that as "no measurement" and passes rather than inventing a denominator.
|
|
312
314
|
*/
|
|
313
315
|
export function parseExitContext(text) {
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
// exponential notation, which the session store's own decimal round-trip then rejects on read --
|
|
325
|
-
// so a value in that range would be written into the store and be unreadable forever after, with
|
|
326
|
-
// the gate failing open on a measurement that said the context was full.
|
|
327
|
-
if (c && typeof c === "object" && !Array.isArray(c) && Number.isSafeInteger(c.tokens) && Number.isSafeInteger(c.window) && c.tokens >= 0 && c.window > 0) {
|
|
328
|
-
return { tokens: c.tokens, window: c.window };
|
|
329
|
-
}
|
|
330
|
-
return null;
|
|
316
|
+
const parsed = decisiveExitLine(text);
|
|
317
|
+
if (parsed === null) return null;
|
|
318
|
+
const c = parsed?.context;
|
|
319
|
+
// A window of 0 is not a denominator, and a negative count is not a measurement. SAFE integers
|
|
320
|
+
// specifically: `Number.isInteger` accepts up to ~1.8e308, and anything from 1e21 up stringifies to
|
|
321
|
+
// exponential notation, which the session store's own decimal round-trip then rejects on read --
|
|
322
|
+
// so a value in that range would be written into the store and be unreadable forever after, with
|
|
323
|
+
// the gate failing open on a measurement that said the context was full.
|
|
324
|
+
if (c && typeof c === "object" && !Array.isArray(c) && Number.isSafeInteger(c.tokens) && Number.isSafeInteger(c.window) && c.tokens >= 0 && c.window > 0) {
|
|
325
|
+
return { tokens: c.tokens, window: c.window };
|
|
331
326
|
}
|
|
332
327
|
return null;
|
|
333
328
|
}
|
|
334
329
|
|
|
330
|
+
/**
|
|
331
|
+
* The keys of the runner's `resources` block (issue #596), in its emission order. A copy of the runner's
|
|
332
|
+
* `RESOURCE_KEYS` (image/runner/src/cgroup-usage.mjs), which the worker cannot import at run time; a test holds the
|
|
333
|
+
* two lists equal.
|
|
334
|
+
*/
|
|
335
|
+
export const RESOURCE_KEYS = Object.freeze(["memPeak", "swapPeak", "oomKills", "memSomeUsec", "memFullUsec", "cpuUsec", "throttledUsec", "throttled", "pidsPeak"]);
|
|
336
|
+
|
|
337
|
+
/**
|
|
338
|
+
* What the job's container used, off the decisive exit line (`decisiveExitLine`, issue #596): `{ memPeak, swapPeak,
|
|
339
|
+
* oomKills, memSomeUsec, memFullUsec, cpuUsec, throttledUsec, throttled, pidsPeak }`, each a safe non-negative integer
|
|
340
|
+
* or null, or null.
|
|
341
|
+
*
|
|
342
|
+
* REBUILT as an explicit literal over RESOURCE_KEYS, never passed through: extra keys are dropped, and a key the
|
|
343
|
+
* runner could not read arrives as null (or absent) and stays null. A present value that is not a safe non-negative
|
|
344
|
+
* integer (a string, a float, a negative, anything past 2^53) nulls the WHOLE block, the malformed->null rule
|
|
345
|
+
* `parseExitUsage` follows: such a line was not written by a conformant runner, and half of a forged block is not a
|
|
346
|
+
* measurement. Null when the line has no `resources` (an image from before this field, a runner that could read no
|
|
347
|
+
* cgroup file) or when every key is null. Read off the line `decisiveExitLine` picks, like every sibling, and NEVER
|
|
348
|
+
* throws. Telemetry, with ONE exception: `oomKills` and `memPeak` gate the OOM classification (`parseExitOomKilled`
|
|
349
|
+
* needs `oomKills` above 0, and the processor needs `memPeak` at 90% of the container's memory limit or more). These
|
|
350
|
+
* numbers are produced inside the job's container, so a job can inflate them; nothing else reads them for a decision.
|
|
351
|
+
*/
|
|
352
|
+
export function parseExitResources(text) {
|
|
353
|
+
const parsed = decisiveExitLine(text);
|
|
354
|
+
if (parsed === null) return null;
|
|
355
|
+
return rebuildResources(parsed?.resources);
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/** The reason the image's supervisor writes when the runner was killed for memory (image/runner/supervise.mjs). */
|
|
359
|
+
export const EXIT_OOM_KILLED = "oom-killed";
|
|
360
|
+
|
|
361
|
+
/**
|
|
362
|
+
* Whether the decisive exit line (`decisiveExitLine`) is the supervisor's report that the runner was killed for memory
|
|
363
|
+
* (issue #596): `by: "supervisor"`, `code: 137`, `reason: "oom-killed"`, and a `resources` block whose `oomKills` is
|
|
364
|
+
* above 0, all on that one line. False for anything else, including a line with no resources, and a supervisor's line
|
|
365
|
+
* that follows one the runner wrote (that runner finished and was killed afterwards). NEVER throws.
|
|
366
|
+
*
|
|
367
|
+
* The processor reads it only beside a container exit of 137 that the worker did not cause, only from a line the
|
|
368
|
+
* per-job key verified (the supervisor runs in images that declare `exitAuth`, so an unsigned line saying this was
|
|
369
|
+
* written by a job's own tool), and only when the line's `memPeak` reached 90% of the container's memory limit
|
|
370
|
+
* (`oom_kill` also counts a kill by the HOST's OOM killer, which a job at 1000 meets first). A child killed for memory
|
|
371
|
+
* while the runner lives writes no such line (the runner ends on its own code), so it never reads as one.
|
|
372
|
+
*/
|
|
373
|
+
export function parseExitOomKilled(text) {
|
|
374
|
+
const parsed = decisiveExitLine(text);
|
|
375
|
+
if (parsed === null) return false;
|
|
376
|
+
const used = rebuildResources(parsed?.resources);
|
|
377
|
+
return parsed?.by === EXIT_BY_SUPERVISOR && parsed?.code === 137 && parsed?.reason === EXIT_OOM_KILLED && used !== null && used.oomKills !== null && used.oomKills > 0;
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
/** The validating rebuild behind `parseExitResources` and the record's `resources`: null on any violation. */
|
|
381
|
+
function rebuildResources(r) {
|
|
382
|
+
if (r === null || typeof r !== "object" || Array.isArray(r)) return null;
|
|
383
|
+
const out = {};
|
|
384
|
+
for (const key of RESOURCE_KEYS) {
|
|
385
|
+
const v = Object.hasOwn(r, key) ? r[key] : null;
|
|
386
|
+
if (v === null || v === undefined) {
|
|
387
|
+
out[key] = null;
|
|
388
|
+
continue;
|
|
389
|
+
}
|
|
390
|
+
if (!Number.isSafeInteger(v) || v < 0) return null;
|
|
391
|
+
out[key] = v;
|
|
392
|
+
}
|
|
393
|
+
return RESOURCE_KEYS.some((key) => out[key] !== null) ? out : null;
|
|
394
|
+
}
|
|
395
|
+
|
|
335
396
|
/**
|
|
336
397
|
* The keys the runner actually emits, in its own emission order: the metered snapshot
|
|
337
398
|
* (`image/runner/src/usage-meter.mjs` -> `snapshot`) plus the token-budget fallback
|
|
@@ -393,17 +454,10 @@ function rebuildTokens(t) {
|
|
|
393
454
|
}
|
|
394
455
|
|
|
395
456
|
export function parseExitTokens(text) {
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
if (line === "") continue;
|
|
401
|
-
const parsed = parseTailLine(line);
|
|
402
|
-
if (parsed?.event !== "exit") continue;
|
|
403
|
-
const t = parsed?.tokens;
|
|
404
|
-
if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return rebuildTokens(t);
|
|
405
|
-
return null;
|
|
406
|
-
}
|
|
457
|
+
const parsed = decisiveExitLine(text);
|
|
458
|
+
if (parsed === null) return null;
|
|
459
|
+
const t = parsed?.tokens;
|
|
460
|
+
if (t && typeof t === "object" && !Array.isArray(t) && typeof t.total === "number") return rebuildTokens(t);
|
|
407
461
|
return null;
|
|
408
462
|
}
|
|
409
463
|
|
|
@@ -435,16 +489,9 @@ const USAGE_ROW_NUMERIC_KEYS = ["calls", "input", "output", "cacheRead", "cacheW
|
|
|
435
489
|
* classification (INT-RUNNER-EXIT-CODE-PROTOCOL).
|
|
436
490
|
*/
|
|
437
491
|
export function parseExitUsage(text) {
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
const line = lines[i].trim();
|
|
442
|
-
if (line === "") continue;
|
|
443
|
-
const parsed = parseTailLine(line);
|
|
444
|
-
if (parsed?.event !== "exit") continue;
|
|
445
|
-
return rebuildUsage(parsed?.usage);
|
|
446
|
-
}
|
|
447
|
-
return null;
|
|
492
|
+
const parsed = decisiveExitLine(text);
|
|
493
|
+
if (parsed === null) return null;
|
|
494
|
+
return rebuildUsage(parsed?.usage);
|
|
448
495
|
}
|
|
449
496
|
|
|
450
497
|
/** The validating rebuild behind `parseExitUsage`: explicit literals only, null on ANY violation. */
|
|
@@ -522,7 +569,7 @@ function rebuildUsage(u) {
|
|
|
522
569
|
* default to `null` when the outcome does not carry them, so the record shape is stable whether or not
|
|
523
570
|
* the source reports those fields.
|
|
524
571
|
*/
|
|
525
|
-
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null }) {
|
|
572
|
+
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null, size = null }) {
|
|
526
573
|
const data = job.data ?? {};
|
|
527
574
|
const kind = data.kind ?? job.name;
|
|
528
575
|
const source = result ?? error ?? {};
|
|
@@ -658,9 +705,45 @@ export function buildRecord({ job, result, error, startedAt, endedAt, host = nul
|
|
|
658
705
|
// with no portfolio trigger; a confirmed one that wrote none says `plan-absent` (issue #507). Never
|
|
659
706
|
// the plan's weights or its reasons: those are in the allocation audit file, and a reason is agent text.
|
|
660
707
|
plan: planOf(source.plan),
|
|
708
|
+
// What the container used (issue #596, INT-RUN-HISTORY-FILE-CONTRACT): peak memory, OOM kills, memory pressure, CPU
|
|
709
|
+
// time and throttling, peak processes, read by the runner from its own cgroup just before its exit line. Additive,
|
|
710
|
+
// nullable, an explicit literal REBUILT here, TAIL position after `plan` on the same contract. Integers only, so
|
|
711
|
+
// PII-free by construction. A runner killed before its line still has one: the image's supervisor reads the cgroup
|
|
712
|
+
// when it reports the death, so a job killed for memory carries the block that shows it. Null when no exit line
|
|
713
|
+
// carried it: an older image, one whose supervisor died too, or a venue that hides the cgroup. Produced inside the
|
|
714
|
+
// job's container, so a job can inflate every number: advisory (docs/insights.md).
|
|
715
|
+
resources: rebuildResources(source.resources),
|
|
716
|
+
// The size the job was given (issue #596, INT-RUN-HISTORY-FILE-CONTRACT): `{ memMiB, cpuCenti, source }`, resolved
|
|
717
|
+
// at the pickup gate from the limits snapshot and passed in beside `project` (start.mjs `recordRun`), so the record
|
|
718
|
+
// says what the job was promised next to what it used (`resources`). Additive, nullable, an explicit literal REBUILT
|
|
719
|
+
// here (`recordedJobSize`), TAIL position after `resources` on the same contract. Two integers and a fixed word
|
|
720
|
+
// (`project` | `env` | `default`), so PII-free by construction. Present on every record written after the pickup
|
|
721
|
+
// gate, including a refusal there that started no container (it is the size the job WOULD have had); null on a record
|
|
722
|
+
// written before it (the wait gate's refusals) and from a caller that passes none.
|
|
723
|
+
size: recordedJobSize(size),
|
|
724
|
+
// The host budget a never-fits refusal judged the size against (issue #596, phase 2, INT-RUN-HISTORY-FILE-CONTRACT):
|
|
725
|
+
// `{ memMiB, cpuCenti, hostShare }`, so the record names BOTH sizes where the forge comment names neither. Additive,
|
|
726
|
+
// nullable, an explicit literal REBUILT here (`recordedHostBudget`), TAIL position after `size` on the same
|
|
727
|
+
// contract. Each budget an integer, `"off"` for a dimension switched off, or null for one unknown (the OTHER dimension
|
|
728
|
+
// refused: an off or unknown dimension never refuses by itself); `hostShare` an integer or null. Present only on
|
|
729
|
+
// the `job-size-exceeds-host` and `-share` records; null on every other.
|
|
730
|
+
hostBudget: recordedHostBudget(source.hostBudget),
|
|
661
731
|
};
|
|
662
732
|
}
|
|
663
733
|
|
|
734
|
+
/**
|
|
735
|
+
* A refusal's host budget as a record carries it: `{ memMiB, cpuCenti, hostShare }`, else null. A budget is a safe
|
|
736
|
+
* integer, `"off"` (the budget's `Infinity`, or the word itself) or null (unknown); the share a safe integer or null.
|
|
737
|
+
* `"off"` rather than null for a switched-off dimension (gate round 1 of phase 2): null already means unknown,
|
|
738
|
+
* and a record that cannot tell "not limited" from "not known" names neither.
|
|
739
|
+
*/
|
|
740
|
+
export function recordedHostBudget(value) {
|
|
741
|
+
if (value === null || typeof value !== "object" || Array.isArray(value)) return null;
|
|
742
|
+
const int = (v) => (Number.isSafeInteger(v) && v >= 0 ? v : null);
|
|
743
|
+
const budget = (v) => (v === Infinity || v === "off" ? "off" : int(v));
|
|
744
|
+
return { memMiB: budget(value.memMiB), cpuCenti: budget(value.cpuCenti), hostShare: int(value.hostShare) };
|
|
745
|
+
}
|
|
746
|
+
|
|
664
747
|
/** What became of a collected plan. */
|
|
665
748
|
export const PLAN_RECORD_OUTCOMES = Object.freeze(["applied", "duplicate", "refused"]);
|
|
666
749
|
/**
|
|
@@ -749,8 +832,9 @@ const SIGNED_TAIL = /,"auth":"([0-9a-f]{64})"\}$/;
|
|
|
749
832
|
*
|
|
750
833
|
* The worker hands each job a fresh random key on the container's stdin, and the runner signs its exit line with
|
|
751
834
|
* HMAC-SHA256 over the line's unsigned bytes. A job's own tool can write any bytes it likes to the container's stdout,
|
|
752
|
-
* but not that MAC: the key left the pipe before any tool existed and the
|
|
753
|
-
* (the image's exec-only node)
|
|
835
|
+
* but not that MAC: the key left the pipe before any tool existed, and the processes holding it (the supervisor and
|
|
836
|
+
* the runner) are closed to the job's uid on /proc (the image's exec-only node) and on the Node inspector (both start
|
|
837
|
+
* with `--disable-sigusr1`). So when a key was issued, only these lines are the runner's, and every `parseExit*`
|
|
754
838
|
* scanner runs over this text instead of the raw tail: a forged line, before or after the genuine one, is not in it.
|
|
755
839
|
*
|
|
756
840
|
* Each candidate starts at a `{"event":"exit"` anchor, the `parseTailLine` repair's reasoning: those raw bytes cannot
|
|
@@ -853,6 +937,8 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
|
|
|
853
937
|
const session = parseExitSession(exitText);
|
|
854
938
|
const usage = parseExitUsage(exitText);
|
|
855
939
|
const context = parseExitContext(exitText);
|
|
940
|
+
const resources = parseExitResources(exitText);
|
|
941
|
+
const oomKilled = parseExitOomKilled(exitText);
|
|
856
942
|
try {
|
|
857
943
|
if (stream !== null) {
|
|
858
944
|
const s = stream;
|
|
@@ -878,7 +964,8 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
|
|
|
878
964
|
}
|
|
879
965
|
// `exitAuth` only when a key was issued, so a keyless close returns exactly the object it always did. `exitWhy`
|
|
880
966
|
// (issue #507) only when the line named one, for the same reason.
|
|
881
|
-
|
|
967
|
+
// `resources` (issue #596) only when the line carried a block, and `exitOomKilled` only when true, for the same reason.
|
|
968
|
+
return { turns, tokens, session, usage, context, exitReason, exitLineCode, ...(exitWhy !== null ? { exitWhy } : {}), ...(resources !== null ? { resources } : {}), ...(oomKilled ? { exitOomKilled: true } : {}), ...(keyed ? { exitAuth } : {}) };
|
|
882
969
|
}
|
|
883
970
|
|
|
884
971
|
return { write, close };
|
package/src/sandbox-store.mjs
CHANGED
|
@@ -129,7 +129,7 @@ const defaultFs = { chmodSync, chownSync, lstatSync, mkdirSync, readFileSync, re
|
|
|
129
129
|
* nothing was retained -- and on `null` the caller has nothing left to do, because every failure path
|
|
130
130
|
* here removes `jobDir` itself. Retention must never leave debris behind.
|
|
131
131
|
*
|
|
132
|
-
* `prepared.sandbox` is `{ jobId, kind, image, backend, jobUser? }`, stamped by `makePrepareWorkspace` (`jobUser`
|
|
132
|
+
* `prepared.sandbox` is `{ jobId, kind, image, backend, jobUser?, podmanStore?, size? }`, stamped by `makePrepareWorkspace` (`jobUser`
|
|
133
133
|
* only when the processor decided one, issue #341). Absent (a bare construction, a test, an unwired dispatcher)
|
|
134
134
|
* means no retention, which keeps such a caller on exactly
|
|
135
135
|
* the pre-feature path.
|
|
@@ -182,6 +182,10 @@ export function retainJobDir(prepared, { sandboxDir, retentionHours = null, fs =
|
|
|
182
182
|
// other manifest is the shape it always was. A sandbox refuses to open under another store, and the sweep
|
|
183
183
|
// holds the run while the podman it asks uses another, because that podman's `ps` answers empty (measured).
|
|
184
184
|
...(typeof meta.podmanStore === "string" ? { podmanStore: meta.podmanStore } : {}),
|
|
185
|
+
// Issue #596: the run's size (`{ memMiB, cpuCenti, source }`), so a sandbox reopens the run at it. Written only when
|
|
186
|
+
// known, so every other manifest is the shape it always was; a manifest without it reopens at the built-in 4g
|
|
187
|
+
// and 2, the size every run had before sizes existed.
|
|
188
|
+
...(meta.size && typeof meta.size === "object" ? { size: meta.size } : {}),
|
|
185
189
|
workspace: rebaseWorkspace(prepared.workspace, jobDir, dest),
|
|
186
190
|
createdAt: new Date(created).toISOString(),
|
|
187
191
|
// Issue #446: the deadline THIS worker's window gives the run, written down, so an opener whose own
|