@edgehero/pi-dispatch 2.1.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +41 -5
- package/README.md +11 -5
- package/deploy/docker-compose.yml +12 -0
- package/deploy/egress-proxy.conf +28 -3
- package/deploy/pi-dispatch-egress-proxy.container +8 -2
- package/package.json +8 -1
- package/src/allocation.mjs +731 -0
- package/src/backends.mjs +243 -0
- package/src/budget.mjs +40 -4
- package/src/cli.mjs +222 -11
- package/src/config.mjs +126 -5
- package/src/daemon-facts.mjs +3 -0
- package/src/deployment-venue.mjs +1 -0
- package/src/doctor.mjs +2261 -203
- package/src/dollar-budget.mjs +373 -0
- package/src/dollar-fingerprint.mjs +83 -0
- package/src/egress-cli.mjs +316 -0
- package/src/egress-proxy-state.mjs +35 -5
- package/src/egress.mjs +12 -0
- package/src/env-allowlist.mjs +107 -6
- package/src/env-file.mjs +194 -25
- package/src/envelope.mjs +413 -0
- package/src/exit-code.mjs +22 -0
- package/src/fleet-lease.mjs +85 -25
- package/src/get-token.mjs +16 -5
- package/src/git-dirty.mjs +67 -0
- package/src/github-app-setup.mjs +6 -3
- package/src/github-host.mjs +5 -3
- package/src/identity.mjs +2 -1
- package/src/image-preflight.mjs +98 -24
- package/src/image-ref.mjs +37 -0
- package/src/import-pi.mjs +4 -2
- package/src/index.mjs +407 -62
- package/src/init.mjs +18 -0
- package/src/job-id.mjs +26 -3
- package/src/live-probes.mjs +24 -9
- package/src/model-catalog.mjs +297 -0
- package/src/model-endpoints.mjs +649 -0
- package/src/model-ref.mjs +151 -0
- package/src/models-json.mjs +262 -0
- package/src/money.mjs +144 -0
- package/src/octokit-log.mjs +65 -0
- package/src/outbox-plan.mjs +218 -0
- package/src/outbox.mjs +29 -9
- package/src/output-cap.mjs +157 -0
- package/src/pause-windows.mjs +81 -2
- package/src/pi-model-loader.mjs +77 -0
- package/src/podman-stack.mjs +16 -3
- package/src/portfolio-snapshot.mjs +304 -0
- package/src/prepare-local.mjs +247 -12
- package/src/prepare.mjs +35 -3
- package/src/priorities.mjs +569 -0
- package/src/processor.mjs +599 -170
- package/src/project-id.mjs +17 -0
- package/src/projects.mjs +238 -0
- package/src/provider-steering.mjs +179 -65
- package/src/queue.mjs +111 -6
- package/src/reserved-env.mjs +30 -0
- package/src/run-container.mjs +59 -5
- package/src/run-history.mjs +379 -24
- package/src/run-mirror.mjs +30 -0
- package/src/runtime-settings.mjs +104 -9
- package/src/schedules.mjs +33 -1
- package/src/scoped-limits.mjs +447 -27
- package/src/service.mjs +15 -4
- package/src/session-store.mjs +131 -6
- package/src/start.mjs +528 -40
- package/src/triggers-file.mjs +65 -4
- package/src/triggers.mjs +135 -7
- package/src/up.mjs +308 -34
- package/src/valkey-endpoint.mjs +3 -2
package/src/run-history.mjs
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
|
+
import { createHmac, timingSafeEqual } from "node:crypto";
|
|
1
2
|
import * as nodeFs from "node:fs";
|
|
2
3
|
import { scrubCredentials } from "./redact.mjs";
|
|
3
4
|
import { basename, join } from "node:path";
|
|
4
5
|
import { resolveBackendName } from "./backend-registry.mjs";
|
|
5
6
|
import { isForgeKind, targetSeparator } from "./forges.mjs";
|
|
7
|
+
import { MODEL_REF_PATTERN as USAGE_ID_PATTERN } from "./model-ref.mjs";
|
|
8
|
+
import { isProjectId } from "./project-id.mjs";
|
|
6
9
|
|
|
7
10
|
/**
|
|
8
11
|
* Durable per-run history.
|
|
@@ -118,8 +121,15 @@ export function parseExitTurns(text) {
|
|
|
118
121
|
* below admits nothing else. A test requires a TERMINAL_COMMENTS row per member. The runner writes the
|
|
119
122
|
* literal in `image/runner/src/outcome.mjs`, which the shipped worker cannot import, so a test reads that
|
|
120
123
|
* source and requires every member here to appear there verbatim.
|
|
124
|
+
*
|
|
125
|
+
* The last four are the policy stops of issues #501 (a per-job dollar cap) and #502 (an allowed-model list):
|
|
126
|
+
* `cost-cap` and `model-not-allowed` when the runner's pre-call guard stopped a call, and
|
|
127
|
+
* `cost-cap-unenforceable` and `model-policy-unenforceable` when the runner refused, before any call, a policy
|
|
128
|
+
* it could not enforce before a call. The two refusals are live from the first image that knows the variables.
|
|
129
|
+
* `cost-cap` is live from the image that declares `costCap`, whose runner carries the cost guard, and
|
|
130
|
+
* `model-not-allowed` from the image that declares `modelPolicy`, whose runner carries the model guard.
|
|
121
131
|
*/
|
|
122
|
-
export const RUNNER_POLICY_REASONS = new Set(["provider-auth-refused"]);
|
|
132
|
+
export const RUNNER_POLICY_REASONS = new Set(["provider-auth-refused", "cost-cap", "model-not-allowed", "cost-cap-unenforceable", "model-policy-unenforceable"]);
|
|
123
133
|
|
|
124
134
|
/**
|
|
125
135
|
* The reason off the LAST runner exit line, or `null`: a member of `RUNNER_POLICY_REASONS` and only when
|
|
@@ -147,6 +157,59 @@ export function parseExitReason(text) {
|
|
|
147
157
|
return null;
|
|
148
158
|
}
|
|
149
159
|
|
|
160
|
+
/**
|
|
161
|
+
* The cost guard's refusal rules a `cost-cap` exit line may name as its `why` (issue #507): the runner's
|
|
162
|
+
* `COST_REFUSALS` (`image/runner/src/outcome.mjs`), restated because the shipped worker cannot import the runner, and
|
|
163
|
+
* pinned to it by a test that reads that source. CLOSED, because the value reaches the PII-free record.
|
|
164
|
+
*/
|
|
165
|
+
export const COST_CAP_WHYS = Object.freeze(["unboundable", "external", "over-cap"]);
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* The `why` off the LAST runner exit line, or `null`: a member of `COST_CAP_WHYS`, and only when that same line says
|
|
169
|
+
* `code: 2` and `reason: "cost-cap"`. Scanned from the end exactly as `parseExitReason` is, and NEVER throws.
|
|
170
|
+
*
|
|
171
|
+
* The line is container-written, so nothing it carries is trusted as text: a value outside the closed set is null,
|
|
172
|
+
* never a string copied through. It feeds no classification. The processor reads it only in its exit-2 branch, and
|
|
173
|
+
* only beside a `cost-cap` reason, where it becomes the record's `why`: the worst a forged line can do is name the
|
|
174
|
+
* wrong rule of the three.
|
|
175
|
+
*/
|
|
176
|
+
export function parseExitWhy(text) {
|
|
177
|
+
if (typeof text !== "string") return null;
|
|
178
|
+
const lines = text.split("\n");
|
|
179
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
180
|
+
const line = lines[i].trim();
|
|
181
|
+
if (line === "") continue;
|
|
182
|
+
const parsed = parseTailLine(line);
|
|
183
|
+
if (parsed?.event !== "exit") continue;
|
|
184
|
+
return parsed?.code === 2 && parsed?.reason === "cost-cap" && COST_CAP_WHYS.includes(parsed?.why) ? parsed.why : null;
|
|
185
|
+
}
|
|
186
|
+
return null;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* The LAST runner exit line's own `code`, an integer, or `null` (issue #501, PR #542's review round 3). Scanned from
|
|
191
|
+
* the end exactly as its siblings are, repairing a glued line through `parseTailLine`, and NEVER throws.
|
|
192
|
+
*
|
|
193
|
+
* Read for ONE purpose: the dollar settlement trusts an exit line's cost only when the line's `code` equals the
|
|
194
|
+
* container's real exit code (processor.mjs). The runner writes the code it exits with on both exit lines
|
|
195
|
+
* (`image/runner/run-job.mjs`: `...capExitMessage(outcome)` then `return outcome.code` on the decided path, and
|
|
196
|
+
* `code: capped.code` on the catch path; pinned by a test), so a genuine last line always matches, while a line a
|
|
197
|
+
* job's own tool forged before a `docker stop` (exit 137, no genuine line after it) does not. It never feeds the
|
|
198
|
+
* retry class (INT-RUNNER-EXIT-CODE-PROTOCOL).
|
|
199
|
+
*/
|
|
200
|
+
export function parseExitCode(text) {
|
|
201
|
+
if (typeof text !== "string") return null;
|
|
202
|
+
const lines = text.split("\n");
|
|
203
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
204
|
+
const line = lines[i].trim();
|
|
205
|
+
if (line === "") continue;
|
|
206
|
+
const parsed = parseTailLine(line);
|
|
207
|
+
if (parsed?.event !== "exit") continue;
|
|
208
|
+
return Number.isSafeInteger(parsed?.code) ? parsed.code : null;
|
|
209
|
+
}
|
|
210
|
+
return null;
|
|
211
|
+
}
|
|
212
|
+
|
|
150
213
|
/**
|
|
151
214
|
* Recover the agent's token usage from buffered container stdout, or `null` if it is not reported.
|
|
152
215
|
*
|
|
@@ -186,6 +249,7 @@ export const SESSION_REASONS = new Set([
|
|
|
186
249
|
"conversation-too-old",
|
|
187
250
|
"resume-chain-too-long",
|
|
188
251
|
"context-too-full",
|
|
252
|
+
"compaction-summary-empty",
|
|
189
253
|
"too-large",
|
|
190
254
|
"unparseable",
|
|
191
255
|
"not-a-regular-file",
|
|
@@ -274,8 +338,31 @@ export function parseExitContext(text) {
|
|
|
274
338
|
* (`image/runner/run-job.mjs` -> `pickTotals`, which sends the first four and `metered: false`). Order
|
|
275
339
|
* matters because it is what makes a conformant runner's object round-trip byte-identically through the
|
|
276
340
|
* rebuild below, so the record's bytes do not move for anyone running a real image.
|
|
341
|
+
*
|
|
342
|
+
* After `unpriced` come the three child keys of issue #500 (`DES-USAGE-METER-VIA-API-PROVIDER-REGISTRY`), which a
|
|
343
|
+
* metered runner from part E on writes on every line, zeros with no children: `childTotal` (the children's billed
|
|
344
|
+
* tokens, the fourth part of root + other + loose + child = total), `childProcesses` (child ledger files plus pi
|
|
345
|
+
* processes found with none) and `unmeteredChildren` (pi children whose spend the job could not count). The last is a
|
|
346
|
+
* floor counter (`FLOOR_COUNTERS`), so it is on this list for the same reason as the policy counters below: a dropped
|
|
347
|
+
* key would read as an honest zero. A worker before issue #500 part F drops all three, and the `total` it keeps
|
|
348
|
+
* already includes the children.
|
|
349
|
+
*
|
|
350
|
+
* Then `costUnreported` (issue #571), which the metered runner writes on every line, capped or not, after the child
|
|
351
|
+
* keys: calls on a priced model whose answer carried broken usage (pi records a missing usage block as zeros), so
|
|
352
|
+
* `cost` is short. A floor counter, on this list for the same reason as `unmeteredChildren`.
|
|
353
|
+
*
|
|
354
|
+
* The last seven are the policy counters of issues #501 and #502, in the order the runner emits them after
|
|
355
|
+
* it: `costCapMicros` (the job's cap), `costRefused` (calls the cost guard stopped),
|
|
356
|
+
* `boundExceeded` (calls that cost more than their bound), `longContext` (calls priced past a long-context
|
|
357
|
+
* threshold the catalog does not tier), `costUnjudged` (compat entries found displaced under the cap, so calls
|
|
358
|
+
* may have run unjudged and unmetered), `costUnanswered` (failed calls that never started, charged their metered
|
|
359
|
+
* cost though a provider may have billed one, and since issue #571 calls whose dispatch threw before any answer),
|
|
360
|
+
* `modelRefused` (calls the model guard stopped). The cost guard
|
|
361
|
+
* writes the first six whenever a cost cap is set, and the model guard `modelRefused` whenever a list is. All seven are on
|
|
362
|
+
* this closed list because a key missing from it is DROPPED, and the dollar settlement reads them to decide
|
|
363
|
+
* whether a metered cost is complete: a dropped counter would read as an honest zero.
|
|
277
364
|
*/
|
|
278
|
-
const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced"];
|
|
365
|
+
export const TOKEN_KEYS = Object.freeze(["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced", "childTotal", "childProcesses", "unmeteredChildren", "costUnreported", "costCapMicros", "costRefused", "boundExceeded", "longContext", "costUnjudged", "costUnanswered", "modelRefused"]);
|
|
279
366
|
|
|
280
367
|
/**
|
|
281
368
|
* Rebuild the billed totals from a closed key list rather than passing the container's object through.
|
|
@@ -289,7 +376,7 @@ const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal",
|
|
|
289
376
|
* did not, and the asymmetry was an oversight rather than a decision.
|
|
290
377
|
*
|
|
291
378
|
* A key the runner omitted stays OMITTED rather than becoming null: the fallback shape legitimately
|
|
292
|
-
* carries only five of the
|
|
379
|
+
* carries only five of the twenty-three, and a null there would read as "measured zero" for a number nobody
|
|
293
380
|
* measured. `typeof === "number"` rather than `Number.isFinite`, deliberately, so this narrows WHICH
|
|
294
381
|
* KEYS survive and never which objects are admitted -- the admission gate above is unchanged.
|
|
295
382
|
*/
|
|
@@ -320,10 +407,6 @@ export function parseExitTokens(text) {
|
|
|
320
407
|
return null;
|
|
321
408
|
}
|
|
322
409
|
|
|
323
|
-
/** The id allowlist for a ledger row's provider/model, applied AFTER lowercasing. The first-char class
|
|
324
|
-
* has no dot, colon or slash, so `.hidden`, `../etc` and `:` shapes fail at character one. */
|
|
325
|
-
const USAGE_ID_PATTERN = /^[a-z0-9][a-z0-9._:/-]{0,63}$/;
|
|
326
|
-
|
|
327
410
|
/** The ten per-row counters, in the row's serialisation order. Absent is an honest zero; anything
|
|
328
411
|
* present must be a finite non-negative number or the whole block is refused. */
|
|
329
412
|
const USAGE_ROW_NUMERIC_KEYS = ["calls", "input", "output", "cacheRead", "cacheWrite", "cacheWrite1h", "reasoning", "total", "cost", "unpriced"];
|
|
@@ -376,7 +459,10 @@ function rebuildUsage(u) {
|
|
|
376
459
|
if (typeof u.piAi !== "string" || !/^\d+\.\d+\.\d+$/.test(u.piAi)) return null;
|
|
377
460
|
piAi = u.piAi;
|
|
378
461
|
}
|
|
379
|
-
|
|
462
|
+
// ABSENT stays null, never 0 (PR #549's review): a runner that did not report how many rows it folded did not
|
|
463
|
+
// measure it, and the per-model dollar settlement floors on anything but a present 0. Readers that count folded
|
|
464
|
+
// ledgers treat null as none (`?? 0`).
|
|
465
|
+
let truncated = null;
|
|
380
466
|
if (u.truncated !== undefined) {
|
|
381
467
|
if (!Number.isInteger(u.truncated) || u.truncated < 0) return null;
|
|
382
468
|
truncated = u.truncated;
|
|
@@ -392,6 +478,7 @@ function rebuildUsage(u) {
|
|
|
392
478
|
// itself never has to admit uppercase.
|
|
393
479
|
const provider = row.provider.toLowerCase();
|
|
394
480
|
const model = row.model.toLowerCase();
|
|
481
|
+
// The id allowlist (model-ref.mjs, widened for issues #501/#502 so every builtin catalog id passes).
|
|
395
482
|
if (!USAGE_ID_PATTERN.test(provider) || !USAGE_ID_PATTERN.test(model)) return null;
|
|
396
483
|
const nums = {};
|
|
397
484
|
for (const key of USAGE_ROW_NUMERIC_KEYS) {
|
|
@@ -430,12 +517,12 @@ function rebuildUsage(u) {
|
|
|
430
517
|
* path stay out of the record -- for local jobs only the folder's `basename` is kept, because the full
|
|
431
518
|
* path embeds the operator's OS account name.
|
|
432
519
|
*
|
|
433
|
-
* `reason` is a fixed enum passthrough (worker-abort | over-budget | unprotected-branch |
|
|
520
|
+
* `reason` is a fixed enum passthrough (worker-abort | over-budget | dollar-cap | allocation-cap | envelope-mismatch | portfolio-no-envelope | portfolio-snapshot-oversize | unprotected-branch |
|
|
434
521
|
* runner-policy | provider-auth-refused | job-image-missing | egress-proxy-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
|
|
435
522
|
* default to `null` when the outcome does not carry them, so the record shape is stable whether or not
|
|
436
523
|
* the source reports those fields.
|
|
437
524
|
*/
|
|
438
|
-
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null }) {
|
|
525
|
+
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null }) {
|
|
439
526
|
const data = job.data ?? {};
|
|
440
527
|
const kind = data.kind ?? job.name;
|
|
441
528
|
const source = result ?? error ?? {};
|
|
@@ -545,9 +632,85 @@ export function buildRecord({ job, result, error, startedAt, endedAt, host = nul
|
|
|
545
632
|
// default was passed (a dependency-injection seam; `recordRun` in start.mjs always passes one) or
|
|
546
633
|
// where the job data carries an explicit null, which no producer writes and the blessed gate refuses.
|
|
547
634
|
backend: resolveBackendName(data, defaultBackend),
|
|
635
|
+
// The dollar reservation's outcome (issue #501, INT-RUN-HISTORY-FILE-CONTRACT). Additive, nullable, an explicit
|
|
636
|
+
// literal REBUILT here (never the source's object), TAIL position after `backend` on the same contract: field order
|
|
637
|
+
// is the serialisation order. `{ reservedMicros, settledMicros, basis, modelBasis }`: two integers of micro-dollars,
|
|
638
|
+
// one fixed token (`metered` | `floor` | `refunded` | `unreserved`), and `modelBasis`, how the per-model windows
|
|
639
|
+
// settled (`metered` | `floor` | `refunded`, null when the job held none). Null when no dollar window applied to the job (no dollar setting, or a
|
|
640
|
+
// refusal before the reservation step), which is every record of a deployment that sets none.
|
|
641
|
+
dollars: dollarsOf(source.dollars),
|
|
642
|
+
// The refusal's detail (PR #558, the end-of-round check of #501 and #502): which rule of the reason refused the job, the same fixed token
|
|
643
|
+
// the worker's log line carries, e.g. `overlay-link` under `model-unknown`, so the drill-in can tell the
|
|
644
|
+
// deployment's file from the job's model. Additive, nullable, an explicit literal, TAIL position after `dollars`
|
|
645
|
+
// on the same contract. A token of the fixed charset (`WHY_RE`) or null, never a free string, so the record stays
|
|
646
|
+
// PII-free by construction; null for every reason that carries no detail.
|
|
647
|
+
why: typeof source.why === "string" && WHY_RE.test(source.why) ? source.why : null,
|
|
648
|
+
// The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT). Additive, nullable, an explicit literal, TAIL position
|
|
649
|
+
// after `why` on the same contract. The project's ID only, never its `name`: the id is operator-authored and
|
|
650
|
+
// charset-checked (`PROJECT_ID_RE`), never payload, the argument `host` and `backend` make, so the record stays
|
|
651
|
+
// PII-free by construction; the name is free text and has no path here. Passed in, resolved at the pickup gate
|
|
652
|
+
// (start.mjs `recordRun`), so `buildRecord` stays pure. Anything that is not a well-formed id records null.
|
|
653
|
+
project: isProjectId(project) ? project : null,
|
|
654
|
+
// The priorities plan a completed portfolio job wrote (issue #505, INT-RUN-HISTORY-FILE-CONTRACT). Additive, nullable,
|
|
655
|
+
// an explicit literal REBUILT here, TAIL position after `project` on the same contract. `{ outcome, reason, planId,
|
|
656
|
+
// clamped }`: a fixed outcome, a fixed reason or null, a 16-hex content hash or null, a boolean. Null for every
|
|
657
|
+
// run that left no `/outbox/priorities.json` and was no confirmed portfolio job, which is every run of a deployment
|
|
658
|
+
// with no portfolio trigger; a confirmed one that wrote none says `plan-absent` (issue #507). Never
|
|
659
|
+
// the plan's weights or its reasons: those are in the allocation audit file, and a reason is agent text.
|
|
660
|
+
plan: planOf(source.plan),
|
|
548
661
|
};
|
|
549
662
|
}
|
|
550
663
|
|
|
664
|
+
/** What became of a collected plan. */
|
|
665
|
+
export const PLAN_RECORD_OUTCOMES = Object.freeze(["applied", "duplicate", "refused"]);
|
|
666
|
+
/**
|
|
667
|
+
* Every reason a record's `plan` may carry: the collector's own rungs (outbox-plan.mjs `PLAN_COLLECT_REASONS`), the
|
|
668
|
+
* plan's judgement (`plan-invalid`) and the apply ladder (priorities.mjs `PLAN_LADDER`, with `envelope-mismatch`).
|
|
669
|
+
* Restated here so this module imports neither, and pinned to them by a test.
|
|
670
|
+
*/
|
|
671
|
+
export const PLAN_RECORD_REASONS = Object.freeze([
|
|
672
|
+
"plan-absent",
|
|
673
|
+
"plan-not-portfolio",
|
|
674
|
+
"plan-oversize",
|
|
675
|
+
"plan-not-regular-file",
|
|
676
|
+
"plan-unreadable",
|
|
677
|
+
"plan-parse-error",
|
|
678
|
+
"plan-collect-error",
|
|
679
|
+
"plan-invalid",
|
|
680
|
+
"delegation-off",
|
|
681
|
+
"writer-not-allowed",
|
|
682
|
+
"envelope-mismatch",
|
|
683
|
+
"plan-duplicate",
|
|
684
|
+
"plan-stale",
|
|
685
|
+
"plan-too-soon",
|
|
686
|
+
"plan-incomplete",
|
|
687
|
+
"plan-busy",
|
|
688
|
+
]);
|
|
689
|
+
const PLAN_ID_RE = /^[0-9a-f]{16}$/;
|
|
690
|
+
|
|
691
|
+
/** The record's `plan`, rebuilt from named fields, or null when the source carries none or a malformed one. */
|
|
692
|
+
function planOf(p) {
|
|
693
|
+
if (p === null || typeof p !== "object" || !PLAN_RECORD_OUTCOMES.includes(p.outcome)) return null;
|
|
694
|
+
const reason = PLAN_RECORD_REASONS.includes(p.reason) ? p.reason : null;
|
|
695
|
+
// An applied plan has no reason, and a duplicate or a refusal always has one; a source that breaks that is not trusted.
|
|
696
|
+
if ((p.outcome === "applied") !== (reason === null)) return null;
|
|
697
|
+
return { outcome: p.outcome, reason, planId: typeof p.planId === "string" && PLAN_ID_RE.test(p.planId) ? p.planId : null, clamped: p.clamped === true };
|
|
698
|
+
}
|
|
699
|
+
|
|
700
|
+
/** The charset a record's `why` must match: a lowercase token, the shape of every `why` the processor returns. */
|
|
701
|
+
const WHY_RE = /^[a-z][a-z0-9-]{0,63}$/;
|
|
702
|
+
|
|
703
|
+
const DOLLAR_BASES = new Set(["metered", "floor", "refunded", "unreserved"]);
|
|
704
|
+
// How the job's MODEL windows settled (issue #502 part 6); anything else, null included, records null.
|
|
705
|
+
const MODEL_BASES = new Set(["metered", "floor", "refunded"]);
|
|
706
|
+
|
|
707
|
+
/** The record's `dollars`, rebuilt from named fields only, or null when the source carries none or a malformed one. */
|
|
708
|
+
function dollarsOf(d) {
|
|
709
|
+
if (d === null || typeof d !== "object") return null;
|
|
710
|
+
if (!Number.isSafeInteger(d.reservedMicros) || d.reservedMicros < 0 || !Number.isSafeInteger(d.settledMicros) || d.settledMicros < 0 || !DOLLAR_BASES.has(d.basis)) return null;
|
|
711
|
+
return { reservedMicros: d.reservedMicros, settledMicros: d.settledMicros, basis: d.basis, modelBasis: MODEL_BASES.has(d.modelBasis) ? d.modelBasis : null };
|
|
712
|
+
}
|
|
713
|
+
|
|
551
714
|
/**
|
|
552
715
|
* A stable, non-PII target label. Forge jobs read `repo<sep>number`; local jobs read `local:<basename>` --
|
|
553
716
|
* basename only, so the full folder path (which on Windows carries the OS account name) never lands in
|
|
@@ -577,6 +740,49 @@ export function targetFor(kind, data) {
|
|
|
577
740
|
/** Retain only the last ~8KB of container output for turn recovery, so per-job memory stays flat. */
|
|
578
741
|
const TAIL_CAP_BYTES = 8 * 1024;
|
|
579
742
|
|
|
743
|
+
/** A signed exit line's tail: `,"auth":"<64 hex>"}`, the MAC as the LAST key (image/runner/src/exit-line.mjs). */
|
|
744
|
+
const SIGNED_TAIL = /,"auth":"([0-9a-f]{64})"\}$/;
|
|
745
|
+
|
|
746
|
+
/**
|
|
747
|
+
* The runner exit lines in `text` that carry a valid MAC under `key`, each with its `auth` key removed, joined by
|
|
748
|
+
* newlines in their original order; "" when there is none (issue #545, INT-RUNNER-EXIT-CODE-PROTOCOL).
|
|
749
|
+
*
|
|
750
|
+
* The worker hands each job a fresh random key on the container's stdin, and the runner signs its exit line with
|
|
751
|
+
* HMAC-SHA256 over the line's unsigned bytes. A job's own tool can write any bytes it likes to the container's stdout,
|
|
752
|
+
* but not that MAC: the key left the pipe before any tool existed and the runner's memory is closed to the job's uid
|
|
753
|
+
* (the image's exec-only node). So when a key was issued, only these lines are the runner's, and every `parseExit*`
|
|
754
|
+
* scanner runs over this text instead of the raw tail: a forged line, before or after the genuine one, is not in it.
|
|
755
|
+
*
|
|
756
|
+
* Each candidate starts at a `{"event":"exit"` anchor, the `parseTailLine` repair's reasoning: those raw bytes cannot
|
|
757
|
+
* occur inside a runner line, so a line glued to stray bytes before it is still found, and an anchor inside the stray
|
|
758
|
+
* bytes simply fails the MAC. The comparison is `timingSafeEqual` on equal-length digests. Never throws.
|
|
759
|
+
*/
|
|
760
|
+
export function authenticExitLines(text, key) {
|
|
761
|
+
if (typeof text !== "string" || typeof key !== "string" || key === "") return "";
|
|
762
|
+
const verified = [];
|
|
763
|
+
for (const raw of text.split("\n")) {
|
|
764
|
+
const line = raw.trim();
|
|
765
|
+
const match = SIGNED_TAIL.exec(line);
|
|
766
|
+
if (match === null) continue;
|
|
767
|
+
const signed = line.slice(0, match.index);
|
|
768
|
+
let from = signed.indexOf('{"event":"exit"');
|
|
769
|
+
while (from !== -1) {
|
|
770
|
+
const body = `${signed.slice(from)}}`;
|
|
771
|
+
try {
|
|
772
|
+
const expected = createHmac("sha256", key).update(body, "utf8").digest();
|
|
773
|
+
if (timingSafeEqual(expected, Buffer.from(match[1], "hex"))) {
|
|
774
|
+
verified.push(body);
|
|
775
|
+
break;
|
|
776
|
+
}
|
|
777
|
+
} catch {
|
|
778
|
+
// a digest that will not compare is not a match
|
|
779
|
+
}
|
|
780
|
+
from = signed.indexOf('{"event":"exit"', from + 1);
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
return verified.join("\n");
|
|
784
|
+
}
|
|
785
|
+
|
|
580
786
|
/**
|
|
581
787
|
* The durable log sink: the I/O layer that streams a job's raw container output to a per-job `.log`
|
|
582
788
|
* file and recovers the turn count from a bounded tail of that same output.
|
|
@@ -604,7 +810,10 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
|
|
|
604
810
|
log("logs_dir_error", { reason: err?.message });
|
|
605
811
|
}
|
|
606
812
|
|
|
607
|
-
|
|
813
|
+
// `exitKey` (issue #545): the key this job's runner signs its exit line with, or null when none was issued (an image
|
|
814
|
+
// that does not declare `exitAuth`, or a caller that predates the field). With a key, the exit-line fields are read
|
|
815
|
+
// from the AUTHENTICATED lines only, and none at all reads exactly like a container that wrote no exit line.
|
|
816
|
+
return function openJobLog(jobId, { exitKey = null } = {}) {
|
|
608
817
|
let tail = "";
|
|
609
818
|
let stream = null;
|
|
610
819
|
|
|
@@ -630,12 +839,20 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
|
|
|
630
839
|
// The tail accumulates whether or not `enabled` is set, which is what keeps exitReason in the record
|
|
631
840
|
// when raw logs are off: a label that vanished with log capture would make the record depend on an
|
|
632
841
|
// opt-in PII switch.
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
const
|
|
637
|
-
const
|
|
638
|
-
const
|
|
842
|
+
// With a key, the scanners see only the lines the runner signed (`authenticExitLines`), so a line a job's tool
|
|
843
|
+
// wrote, before or after the genuine one, is never the one they read. `exitAuth` says which case this was:
|
|
844
|
+
// absent (no key issued: today's whole-tail read), "verified" or "unverified" (a key, and no line carried it).
|
|
845
|
+
const keyed = typeof exitKey === "string" && exitKey !== "";
|
|
846
|
+
const exitText = keyed ? authenticExitLines(tail, exitKey) : tail;
|
|
847
|
+
const exitAuth = exitText === "" ? "unverified" : "verified";
|
|
848
|
+
const exitReason = parseExitReason(exitText);
|
|
849
|
+
const exitLineCode = parseExitCode(exitText);
|
|
850
|
+
const exitWhy = parseExitWhy(exitText);
|
|
851
|
+
const turns = parseExitTurns(exitText);
|
|
852
|
+
const tokens = parseExitTokens(exitText);
|
|
853
|
+
const session = parseExitSession(exitText);
|
|
854
|
+
const usage = parseExitUsage(exitText);
|
|
855
|
+
const context = parseExitContext(exitText);
|
|
639
856
|
try {
|
|
640
857
|
if (stream !== null) {
|
|
641
858
|
const s = stream;
|
|
@@ -659,7 +876,9 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
|
|
|
659
876
|
} catch (err) {
|
|
660
877
|
log("log_sink_error", { jobId, reason: err?.message });
|
|
661
878
|
}
|
|
662
|
-
|
|
879
|
+
// `exitAuth` only when a key was issued, so a keyless close returns exactly the object it always did. `exitWhy`
|
|
880
|
+
// (issue #507) only when the line named one, for the same reason.
|
|
881
|
+
return { turns, tokens, session, usage, context, exitReason, exitLineCode, ...(exitWhy !== null ? { exitWhy } : {}), ...(keyed ? { exitAuth } : {}) };
|
|
663
882
|
}
|
|
664
883
|
|
|
665
884
|
return { write, close };
|
|
@@ -716,6 +935,11 @@ export function makeRecordWriter({ logsDir, fs = nodeFs, log = () => {} }) {
|
|
|
716
935
|
* (schedulers "a" vs "a_1": `repeat_a_1_100.json` has a non-digit tail after "repeat_a_", so it never
|
|
717
936
|
* matches scheduler "a"). The max millis strictly below `beforeMillis` is the previous fire.
|
|
718
937
|
*
|
|
938
|
+
* A fire by hand (`pi-dispatch run --trigger <id>`, issue #505) counts as a run of that trigger: its id is
|
|
939
|
+
* `manual:<schedulerId>:<millis>`, the millis floored to the minute it was queued, stored as `manual_<id>_<millis>.json`.
|
|
940
|
+
* It ran the trigger's own data, so the next tick's `previousRunAt` names it rather than an older scheduled fire. A hand
|
|
941
|
+
* fire and a tick in the same minute share a millis; then the one that ended later is the previous run.
|
|
942
|
+
*
|
|
719
943
|
* Returns `record.endedAt ?? record.startedAt ?? null` as an ISO string. `endedAt` first: BullMQ never
|
|
720
944
|
* overlaps two fires of one scheduler, so the prior run's end is the honest high-water mark; `startedAt`
|
|
721
945
|
* covers a crashed run's partial record. ANY failure -- missing dir, no prior run, unreadable file, bad
|
|
@@ -729,26 +953,157 @@ export function makeFindPreviousRun({ logsDir, fs = nodeFs }) {
|
|
|
729
953
|
return function findPreviousRun({ schedulerId, beforeMillis } = {}) {
|
|
730
954
|
try {
|
|
731
955
|
if (typeof beforeMillis !== "number" || !Number.isFinite(beforeMillis)) return null;
|
|
732
|
-
const
|
|
733
|
-
const
|
|
734
|
-
const pattern = new RegExp(`^${escaped}(\\d+)\\.json$`);
|
|
956
|
+
const escape = (text) => text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
957
|
+
const pattern = new RegExp(`^(?:${escape(sanitizeJobId(`repeat:${schedulerId}:`))}|${escape(sanitizeJobId(`manual:${schedulerId}:`))})(\\d+)\\.json$`);
|
|
735
958
|
let best = null;
|
|
736
959
|
for (const name of fs.readdirSync(logsDir)) {
|
|
737
960
|
const m = pattern.exec(name);
|
|
738
961
|
if (m === null) continue;
|
|
739
962
|
const millis = Number(m[1]);
|
|
740
963
|
if (!Number.isFinite(millis) || millis >= beforeMillis) continue;
|
|
741
|
-
if (best === null || millis > best.millis) best = { millis, name };
|
|
964
|
+
if (best === null || millis > best.millis) best = { millis, names: [name] };
|
|
965
|
+
else if (millis === best.millis) best.names.push(name);
|
|
742
966
|
}
|
|
743
967
|
if (best === null) return null;
|
|
744
|
-
|
|
745
|
-
|
|
968
|
+
let at = null;
|
|
969
|
+
for (const name of best.names) {
|
|
970
|
+
const record = JSON.parse(fs.readFileSync(join(logsDir, name), "utf8"));
|
|
971
|
+
const end = record.endedAt ?? record.startedAt ?? null;
|
|
972
|
+
if (typeof end === "string" && (at === null || end > at)) at = end;
|
|
973
|
+
}
|
|
974
|
+
return at;
|
|
746
975
|
} catch {
|
|
747
976
|
return null; // NEVER throws: a history fault must not fail the prepare that asked
|
|
748
977
|
}
|
|
749
978
|
};
|
|
750
979
|
}
|
|
751
980
|
|
|
981
|
+
/**
|
|
982
|
+
* What `makeReadRecord` returns for a record file that exists but cannot be read as a record (unreadable, bad JSON,
|
|
983
|
+
* not an object). Distinct from `null` (no file), so the lost-lock check can say which it met.
|
|
984
|
+
*/
|
|
985
|
+
export const UNREADABLE_RECORD = Object.freeze({ unreadable: true });
|
|
986
|
+
|
|
987
|
+
/**
|
|
988
|
+
* Read one job's run record back, by the name `makeRecordWriter` gave it: `<logsDir>/<sanitizeJobId(id)>.json`
|
|
989
|
+
* (INT-RUN-HISTORY-FILE-CONTRACT, unchanged). Returns the parsed object, `null` when there is no file, or
|
|
990
|
+
* `UNREADABLE_RECORD` when there is one that is not a record.
|
|
991
|
+
*
|
|
992
|
+
* Why the worker reads its own record: the queue can lose a job's lock after the processor finished and wrote
|
|
993
|
+
* this file, when Valkey was unreachable for longer than BullMQ's lock renewal window. BullMQ then refuses the
|
|
994
|
+
* completion ("Missing lock") and its stall check takes the job back, so the queue alone says the job failed or
|
|
995
|
+
* must run again. The record is the store that already knows it finished (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK).
|
|
996
|
+
*
|
|
997
|
+
* The id inside is NOT checked here: `sanitizeJobId` maps `a:b` and `a_b` to one name, and `recordVerdict` refuses
|
|
998
|
+
* the other job's record as `id-mismatch`, which is the one place that can also say so. NEVER throws.
|
|
999
|
+
*/
|
|
1000
|
+
export function makeReadRecord({ logsDir, fs = nodeFs }) {
|
|
1001
|
+
return function readRecord(jobId) {
|
|
1002
|
+
if (jobId === null || jobId === undefined || jobId === "") return null;
|
|
1003
|
+
let text;
|
|
1004
|
+
try {
|
|
1005
|
+
text = fs.readFileSync(join(logsDir, `${sanitizeJobId(jobId)}.json`), "utf8");
|
|
1006
|
+
} catch (err) {
|
|
1007
|
+
return err?.code === "ENOENT" || err?.code === "ENOTDIR" ? null : UNREADABLE_RECORD;
|
|
1008
|
+
}
|
|
1009
|
+
try {
|
|
1010
|
+
const record = JSON.parse(text);
|
|
1011
|
+
return record !== null && typeof record === "object" && !Array.isArray(record) ? record : UNREADABLE_RECORD;
|
|
1012
|
+
} catch {
|
|
1013
|
+
return UNREADABLE_RECORD;
|
|
1014
|
+
}
|
|
1015
|
+
};
|
|
1016
|
+
}
|
|
1017
|
+
|
|
1018
|
+
/**
|
|
1019
|
+
* How far a record's `startedAt` may sit BEFORE the job's creation and still be this job's record: 5 minutes.
|
|
1020
|
+
*
|
|
1021
|
+
* The two instants come from two clocks. `job.timestamp` is stamped by whoever ENQUEUED the job (the receiver, the
|
|
1022
|
+
* CLI, another worker) and `startedAt` by the worker that ran it, so a producer whose clock runs ahead of the
|
|
1023
|
+
* worker's would make every genuine record look older than its job and silently turn this check off. The
|
|
1024
|
+
* tolerance absorbs ordinary skew between hosts; what it must still refuse is a record of an OLDER job under a
|
|
1025
|
+
* reused id, and such a record is older by at least that job's whole run, its removal from the queue and a new
|
|
1026
|
+
* enqueue, which in practice is far more than 5 minutes. A skew larger than this is an operator problem the
|
|
1027
|
+
* `job_lost_lock_record_rejected` line (`older-than-job`) makes visible.
|
|
1028
|
+
*/
|
|
1029
|
+
export const RECORD_CLOCK_SKEW_MS = 5 * 60 * 1000;
|
|
1030
|
+
|
|
1031
|
+
/**
|
|
1032
|
+
* Does this record say that THIS attempt of THIS job finished without failing? Returns `null` when it does,
|
|
1033
|
+
* `"absent"` when there is no record, and otherwise the reason it does not, from a fixed set:
|
|
1034
|
+
* `unreadable`, `id-mismatch`, `other-attempt`, `failed`, `older-than-job`. Pure; never throws.
|
|
1035
|
+
*
|
|
1036
|
+
* `attempt` is the 1-based attempt number the record would carry, which is BullMQ's `attemptsMade + 1` while the
|
|
1037
|
+
* job is processing (see `buildRecord`) and `attemptsMade` after BullMQ failed it (its moveToFailed adds one
|
|
1038
|
+
* before the failed event). A record of an earlier attempt does not answer for a later one.
|
|
1039
|
+
*
|
|
1040
|
+
* `since` is the job's creation time in millis (BullMQ's `job.timestamp`). A job id can be used again once the
|
|
1041
|
+
* old job is gone (a fixed `--trigger` id, a redelivered webhook), while its record lives for the retention
|
|
1042
|
+
* window; a record that started before this job existed (less `RECORD_CLOCK_SKEW_MS`) belongs to the old one. A
|
|
1043
|
+
* missing or unparseable `startedAt` is not trusted for the same reason.
|
|
1044
|
+
*
|
|
1045
|
+
* `failed` (an outcome of `failed`, or none) is refused on purpose: a record written by the catch path says the run
|
|
1046
|
+
* threw, and the queue's verdict (failed, or run again) is then the right one.
|
|
1047
|
+
*/
|
|
1048
|
+
export function recordVerdict(record, { jobId, attempt, since } = {}) {
|
|
1049
|
+
try {
|
|
1050
|
+
if (record === null || record === undefined) return "absent";
|
|
1051
|
+
if (record === UNREADABLE_RECORD || typeof record !== "object" || Array.isArray(record)) return "unreadable";
|
|
1052
|
+
if (record.jobId !== jobId) return "id-mismatch";
|
|
1053
|
+
if (!Number.isInteger(attempt) || record.attempt !== attempt) return "other-attempt";
|
|
1054
|
+
if (typeof record.outcome !== "string" || record.outcome === "failed") return "failed";
|
|
1055
|
+
if (Number.isFinite(since)) {
|
|
1056
|
+
const started = Date.parse(record.startedAt ?? "");
|
|
1057
|
+
if (!Number.isFinite(started) || started < since - RECORD_CLOCK_SKEW_MS) return "older-than-job";
|
|
1058
|
+
}
|
|
1059
|
+
return null;
|
|
1060
|
+
} catch {
|
|
1061
|
+
return "unreadable";
|
|
1062
|
+
}
|
|
1063
|
+
}
|
|
1064
|
+
|
|
1065
|
+
/** `recordVerdict` as a yes or no. */
|
|
1066
|
+
export const recordSettlesAttempt = (record, opts) => recordVerdict(record, opts) === null;
|
|
1067
|
+
|
|
1068
|
+
/**
|
|
1069
|
+
* The lookup the worker asks when the queue lost a job's lock: the local record first, then the fleet's copy.
|
|
1070
|
+
*
|
|
1071
|
+
* On a fleet without a shared `PI_LOGS_DIR`, the host that meets the stalled job is not always the host that ran
|
|
1072
|
+
* it, so the local file can be absent or hold an older attempt. The run mirror (`run-mirror.mjs`) holds the same
|
|
1073
|
+
* record's bytes, so it is asked when the local file does not settle the attempt. `readMirrored` is `null` on a
|
|
1074
|
+
* deployment that declared no worker name, which has no mirror and is one host.
|
|
1075
|
+
*
|
|
1076
|
+
* `onReject(reason, source)` is told about every record it FOUND and refused (`source` is `local` or `mirror`,
|
|
1077
|
+
* `reason` one of `recordVerdict`'s tokens, never `absent`), so the caller can log why the failure path stood.
|
|
1078
|
+
*
|
|
1079
|
+
* Resolves to the record or `null`, and NEVER rejects: a mirror fault is no record, which fails toward today's
|
|
1080
|
+
* behaviour (a failure comment, or a run).
|
|
1081
|
+
*/
|
|
1082
|
+
export function makeSettledRecord({ readRecord, readMirrored = null }) {
|
|
1083
|
+
return async function settledRecord(jobId, { attempt, since, onReject = () => {} } = {}) {
|
|
1084
|
+
const judge = (record, source) => {
|
|
1085
|
+
const why = recordVerdict(record, { jobId, attempt, since });
|
|
1086
|
+
if (why !== null && why !== "absent") {
|
|
1087
|
+
try {
|
|
1088
|
+
onReject(why, source);
|
|
1089
|
+
} catch {
|
|
1090
|
+
// a reporting fault must not change the verdict
|
|
1091
|
+
}
|
|
1092
|
+
}
|
|
1093
|
+
return why === null;
|
|
1094
|
+
};
|
|
1095
|
+
try {
|
|
1096
|
+
const local = readRecord(jobId);
|
|
1097
|
+
if (judge(local, "local")) return local;
|
|
1098
|
+
if (typeof readMirrored !== "function") return null;
|
|
1099
|
+
const mirrored = await readMirrored(jobId);
|
|
1100
|
+
return judge(mirrored, "mirror") ? mirrored : null;
|
|
1101
|
+
} catch {
|
|
1102
|
+
return null;
|
|
1103
|
+
}
|
|
1104
|
+
};
|
|
1105
|
+
}
|
|
1106
|
+
|
|
752
1107
|
/**
|
|
753
1108
|
* The durable log reaper: an age sweep that deletes `.log` and `.json` history files older than the
|
|
754
1109
|
* retention window, keeping the logs directory bounded. Runs at boot AND on the retention timer since
|
package/src/run-mirror.mjs
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { UNREADABLE_RECORD } from "./run-history.mjs";
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* A fleet-visible copy of the run history (issue #57, Gap 3).
|
|
3
5
|
*
|
|
@@ -186,6 +188,34 @@ export async function readMirroredRuns(redis, { limit = 50, sinceMs = 0, now = (
|
|
|
186
188
|
return { runs, degraded: runs.length >= limit ? "truncated" : "ok" };
|
|
187
189
|
}
|
|
188
190
|
|
|
191
|
+
/**
|
|
192
|
+
* One run's mirrored record, by its sanitized id: the parsed object, `null`, or `UNREADABLE_RECORD`. Never throws,
|
|
193
|
+
* never rejects.
|
|
194
|
+
*
|
|
195
|
+
* The by-id read the lost-lock check needs (`makeSettledRecord` in `run-history.mjs`): one bounded `GET`, not the
|
|
196
|
+
* index scan `readMirroredRuns` does for the panel. Unreachable and absent are `null` (nothing found); a value that
|
|
197
|
+
* is there but is not a record (unparseable, empty, not an object) is `UNREADABLE_RECORD`, exactly as the local
|
|
198
|
+
* reader says it, so the check can log `unreadable` from `mirror`. Neither is ever accepted: the caller keeps
|
|
199
|
+
* today's path.
|
|
200
|
+
*/
|
|
201
|
+
export async function readMirroredRecord(redis, sanitizedJobId, { timeoutMs = OP_TIMEOUT_MS } = {}) {
|
|
202
|
+
if (!redis || !sanitizedJobId) return null;
|
|
203
|
+
try {
|
|
204
|
+
const raw = await bounded(redis.get(runRecordKey(sanitizedJobId)), timeoutMs);
|
|
205
|
+
if (raw === null || raw === undefined) return null;
|
|
206
|
+
if (typeof raw !== "string" || raw === "") return UNREADABLE_RECORD;
|
|
207
|
+
let rec;
|
|
208
|
+
try {
|
|
209
|
+
rec = JSON.parse(raw);
|
|
210
|
+
} catch {
|
|
211
|
+
return UNREADABLE_RECORD;
|
|
212
|
+
}
|
|
213
|
+
return rec !== null && typeof rec === "object" && !Array.isArray(rec) ? rec : UNREADABLE_RECORD;
|
|
214
|
+
} catch {
|
|
215
|
+
return null; // unreachable or timed out: nothing was read
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
189
219
|
/**
|
|
190
220
|
* One list from two sources.
|
|
191
221
|
*
|