@edgehero/pi-dispatch 2.0.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/.env.example +44 -7
  2. package/README.md +14 -6
  3. package/deploy/com.pi-dispatch.worker.plist +1 -1
  4. package/deploy/docker-compose.yml +12 -0
  5. package/deploy/egress-proxy.conf +28 -3
  6. package/deploy/pi-dispatch-egress-proxy.container +8 -2
  7. package/deploy/worker-env-wrapper.cmd +1 -1
  8. package/deploy/worker-env-wrapper.sh +3 -3
  9. package/package.json +9 -2
  10. package/src/allocation.mjs +731 -0
  11. package/src/backends.mjs +243 -0
  12. package/src/budget.mjs +40 -4
  13. package/src/cli.mjs +222 -11
  14. package/src/config.mjs +126 -5
  15. package/src/daemon-facts.mjs +3 -0
  16. package/src/deployment-venue.mjs +1 -0
  17. package/src/doctor.mjs +2316 -183
  18. package/src/dollar-budget.mjs +373 -0
  19. package/src/dollar-fingerprint.mjs +83 -0
  20. package/src/egress-cli.mjs +316 -0
  21. package/src/egress-proxy-state.mjs +35 -5
  22. package/src/egress.mjs +16 -3
  23. package/src/env-allowlist.mjs +142 -18
  24. package/src/env-file.mjs +194 -25
  25. package/src/envelope.mjs +413 -0
  26. package/src/exit-code.mjs +22 -0
  27. package/src/fleet-lease.mjs +85 -25
  28. package/src/get-token.mjs +16 -5
  29. package/src/git-dirty.mjs +67 -0
  30. package/src/github-app-setup.mjs +6 -3
  31. package/src/github-host.mjs +5 -3
  32. package/src/host-pi.mjs +19 -3
  33. package/src/identity.mjs +2 -1
  34. package/src/image-preflight.mjs +98 -24
  35. package/src/image-ref.mjs +37 -0
  36. package/src/import-pi.mjs +4 -2
  37. package/src/index.mjs +407 -62
  38. package/src/init.mjs +18 -0
  39. package/src/job-id.mjs +26 -3
  40. package/src/live-probes.mjs +24 -9
  41. package/src/model-catalog.mjs +297 -0
  42. package/src/model-endpoints.mjs +649 -0
  43. package/src/model-ref.mjs +151 -0
  44. package/src/models-json.mjs +262 -0
  45. package/src/money.mjs +144 -0
  46. package/src/octokit-log.mjs +65 -0
  47. package/src/outbox-plan.mjs +218 -0
  48. package/src/outbox.mjs +29 -9
  49. package/src/output-cap.mjs +157 -0
  50. package/src/packages.mjs +2 -2
  51. package/src/pause-windows.mjs +81 -2
  52. package/src/pi-model-loader.mjs +77 -0
  53. package/src/podman-stack.mjs +16 -3
  54. package/src/portfolio-snapshot.mjs +304 -0
  55. package/src/prepare-local.mjs +247 -12
  56. package/src/prepare.mjs +35 -3
  57. package/src/pricing.mjs +9 -5
  58. package/src/priorities.mjs +569 -0
  59. package/src/processor.mjs +603 -173
  60. package/src/project-id.mjs +17 -0
  61. package/src/projects.mjs +238 -0
  62. package/src/provider-key.mjs +32 -7
  63. package/src/provider-steering.mjs +214 -59
  64. package/src/queue.mjs +111 -6
  65. package/src/reserved-env.mjs +30 -0
  66. package/src/run-container.mjs +59 -5
  67. package/src/run-history.mjs +379 -24
  68. package/src/run-mirror.mjs +30 -0
  69. package/src/runtime-settings.mjs +104 -9
  70. package/src/schedules.mjs +33 -1
  71. package/src/scoped-limits.mjs +447 -27
  72. package/src/secrets.mjs +2 -1
  73. package/src/service.mjs +15 -4
  74. package/src/session-store.mjs +131 -6
  75. package/src/start.mjs +528 -40
  76. package/src/subscriptions.mjs +7 -3
  77. package/src/triggers-file.mjs +65 -4
  78. package/src/triggers.mjs +140 -9
  79. package/src/up.mjs +308 -34
  80. package/src/valkey-endpoint.mjs +3 -2
@@ -1,8 +1,11 @@
1
+ import { createHmac, timingSafeEqual } from "node:crypto";
1
2
  import * as nodeFs from "node:fs";
2
3
  import { scrubCredentials } from "./redact.mjs";
3
4
  import { basename, join } from "node:path";
4
5
  import { resolveBackendName } from "./backend-registry.mjs";
5
6
  import { isForgeKind, targetSeparator } from "./forges.mjs";
7
+ import { MODEL_REF_PATTERN as USAGE_ID_PATTERN } from "./model-ref.mjs";
8
+ import { isProjectId } from "./project-id.mjs";
6
9
 
7
10
  /**
8
11
  * Durable per-run history.
@@ -118,8 +121,15 @@ export function parseExitTurns(text) {
118
121
  * below admits nothing else. A test requires a TERMINAL_COMMENTS row per member. The runner writes the
119
122
  * literal in `image/runner/src/outcome.mjs`, which the shipped worker cannot import, so a test reads that
120
123
  * source and requires every member here to appear there verbatim.
124
+ *
125
+ * The last four are the policy stops of issues #501 (a per-job dollar cap) and #502 (an allowed-model list):
126
+ * `cost-cap` and `model-not-allowed` when the runner's pre-call guard stopped a call, and
127
+ * `cost-cap-unenforceable` and `model-policy-unenforceable` when the runner refused, before any call, a policy
128
+ * it could not enforce before a call. The two refusals are live from the first image that knows the variables.
129
+ * `cost-cap` is live from the image that declares `costCap`, whose runner carries the cost guard, and
130
+ * `model-not-allowed` from the image that declares `modelPolicy`, whose runner carries the model guard.
121
131
  */
122
- export const RUNNER_POLICY_REASONS = new Set(["provider-auth-refused"]);
132
+ export const RUNNER_POLICY_REASONS = new Set(["provider-auth-refused", "cost-cap", "model-not-allowed", "cost-cap-unenforceable", "model-policy-unenforceable"]);
123
133
 
124
134
  /**
125
135
  * The reason off the LAST runner exit line, or `null`: a member of `RUNNER_POLICY_REASONS` and only when
@@ -147,6 +157,59 @@ export function parseExitReason(text) {
147
157
  return null;
148
158
  }
149
159
 
160
+ /**
161
+ * The cost guard's refusal rules a `cost-cap` exit line may name as its `why` (issue #507): the runner's
162
+ * `COST_REFUSALS` (`image/runner/src/outcome.mjs`), restated because the shipped worker cannot import the runner, and
163
+ * pinned to it by a test that reads that source. CLOSED, because the value reaches the PII-free record.
164
+ */
165
+ export const COST_CAP_WHYS = Object.freeze(["unboundable", "external", "over-cap"]);
166
+
167
+ /**
168
+ * The `why` off the LAST runner exit line, or `null`: a member of `COST_CAP_WHYS`, and only when that same line says
169
+ * `code: 2` and `reason: "cost-cap"`. Scanned from the end exactly as `parseExitReason` is, and NEVER throws.
170
+ *
171
+ * The line is container-written, so nothing it carries is trusted as text: a value outside the closed set is null,
172
+ * never a string copied through. It feeds no classification. The processor reads it only in its exit-2 branch, and
173
+ * only beside a `cost-cap` reason, where it becomes the record's `why`: the worst a forged line can do is name the
174
+ * wrong rule of the three.
175
+ */
176
+ export function parseExitWhy(text) {
177
+ if (typeof text !== "string") return null;
178
+ const lines = text.split("\n");
179
+ for (let i = lines.length - 1; i >= 0; i--) {
180
+ const line = lines[i].trim();
181
+ if (line === "") continue;
182
+ const parsed = parseTailLine(line);
183
+ if (parsed?.event !== "exit") continue;
184
+ return parsed?.code === 2 && parsed?.reason === "cost-cap" && COST_CAP_WHYS.includes(parsed?.why) ? parsed.why : null;
185
+ }
186
+ return null;
187
+ }
188
+
189
+ /**
190
+ * The LAST runner exit line's own `code`, an integer, or `null` (issue #501, PR #542's review round 3). Scanned from
191
+ * the end exactly as its siblings are, repairing a glued line through `parseTailLine`, and NEVER throws.
192
+ *
193
+ * Read for ONE purpose: the dollar settlement trusts an exit line's cost only when the line's `code` equals the
194
+ * container's real exit code (processor.mjs). The runner writes the code it exits with on both exit lines
195
+ * (`image/runner/run-job.mjs`: `...capExitMessage(outcome)` then `return outcome.code` on the decided path, and
196
+ * `code: capped.code` on the catch path; pinned by a test), so a genuine last line always matches, while a line a
197
+ * job's own tool forged before a `docker stop` (exit 137, no genuine line after it) does not. It never feeds the
198
+ * retry class (INT-RUNNER-EXIT-CODE-PROTOCOL).
199
+ */
200
+ export function parseExitCode(text) {
201
+ if (typeof text !== "string") return null;
202
+ const lines = text.split("\n");
203
+ for (let i = lines.length - 1; i >= 0; i--) {
204
+ const line = lines[i].trim();
205
+ if (line === "") continue;
206
+ const parsed = parseTailLine(line);
207
+ if (parsed?.event !== "exit") continue;
208
+ return Number.isSafeInteger(parsed?.code) ? parsed.code : null;
209
+ }
210
+ return null;
211
+ }
212
+
150
213
  /**
151
214
  * Recover the agent's token usage from buffered container stdout, or `null` if it is not reported.
152
215
  *
@@ -186,6 +249,7 @@ export const SESSION_REASONS = new Set([
186
249
  "conversation-too-old",
187
250
  "resume-chain-too-long",
188
251
  "context-too-full",
252
+ "compaction-summary-empty",
189
253
  "too-large",
190
254
  "unparseable",
191
255
  "not-a-regular-file",
@@ -274,8 +338,31 @@ export function parseExitContext(text) {
274
338
  * (`image/runner/run-job.mjs` -> `pickTotals`, which sends the first four and `metered: false`). Order
275
339
  * matters because it is what makes a conformant runner's object round-trip byte-identically through the
276
340
  * rebuild below, so the record's bytes do not move for anyone running a real image.
341
+ *
342
+ * After `unpriced` come the three child keys of issue #500 (`DES-USAGE-METER-VIA-API-PROVIDER-REGISTRY`), which a
343
+ * metered runner from part E on writes on every line, zeros with no children: `childTotal` (the children's billed
344
+ * tokens, the fourth part of root + other + loose + child = total), `childProcesses` (child ledger files plus pi
345
+ * processes found with none) and `unmeteredChildren` (pi children whose spend the job could not count). The last is a
346
+ * floor counter (`FLOOR_COUNTERS`), so it is on this list for the same reason as the policy counters below: a dropped
347
+ * key would read as an honest zero. A worker before issue #500 part F drops all three, and the `total` it keeps
348
+ * already includes the children.
349
+ *
350
+ * Then `costUnreported` (issue #571), which the metered runner writes on every line, capped or not, after the child
351
+ * keys: calls on a priced model whose answer carried broken usage (pi records a missing usage block as zeros), so
352
+ * `cost` is short. A floor counter, on this list for the same reason as `unmeteredChildren`.
353
+ *
354
+ * The last seven are the policy counters of issues #501 and #502, in the order the runner emits them after
355
+ * it: `costCapMicros` (the job's cap), `costRefused` (calls the cost guard stopped),
356
+ * `boundExceeded` (calls that cost more than their bound), `longContext` (calls priced past a long-context
357
+ * threshold the catalog does not tier), `costUnjudged` (compat entries found displaced under the cap, so calls
358
+ * may have run unjudged and unmetered), `costUnanswered` (failed calls that never started, charged their metered
359
+ * cost though a provider may have billed one, and since issue #571 calls whose dispatch threw before any answer),
360
+ * `modelRefused` (calls the model guard stopped). The cost guard
361
+ * writes the first six whenever a cost cap is set, and the model guard `modelRefused` whenever a list is. All seven are on
362
+ * this closed list because a key missing from it is DROPPED, and the dollar settlement reads them to decide
363
+ * whether a metered cost is complete: a dropped counter would read as an honest zero.
277
364
  */
278
- const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced"];
365
+ export const TOKEN_KEYS = Object.freeze(["input", "output", "total", "cost", "metered", "rootTotal", "otherTotal", "looseTotal", "sessions", "calls", "unresolved", "unpriced", "childTotal", "childProcesses", "unmeteredChildren", "costUnreported", "costCapMicros", "costRefused", "boundExceeded", "longContext", "costUnjudged", "costUnanswered", "modelRefused"]);
279
366
 
280
367
  /**
281
368
  * Rebuild the billed totals from a closed key list rather than passing the container's object through.
@@ -289,7 +376,7 @@ const TOKEN_KEYS = ["input", "output", "total", "cost", "metered", "rootTotal",
289
376
  * did not, and the asymmetry was an oversight rather than a decision.
290
377
  *
291
378
  * A key the runner omitted stays OMITTED rather than becoming null: the fallback shape legitimately
292
- * carries only five of the twelve, and a null there would read as "measured zero" for a number nobody
379
+ * carries only five of the twenty-three, and a null there would read as "measured zero" for a number nobody
293
380
  * measured. `typeof === "number"` rather than `Number.isFinite`, deliberately, so this narrows WHICH
294
381
  * KEYS survive and never which objects are admitted -- the admission gate above is unchanged.
295
382
  */
@@ -320,10 +407,6 @@ export function parseExitTokens(text) {
320
407
  return null;
321
408
  }
322
409
 
323
- /** The id allowlist for a ledger row's provider/model, applied AFTER lowercasing. The first-char class
324
- * has no dot, colon or slash, so `.hidden`, `../etc` and `:` shapes fail at character one. */
325
- const USAGE_ID_PATTERN = /^[a-z0-9][a-z0-9._:/-]{0,63}$/;
326
-
327
410
  /** The ten per-row counters, in the row's serialisation order. Absent is an honest zero; anything
328
411
  * present must be a finite non-negative number or the whole block is refused. */
329
412
  const USAGE_ROW_NUMERIC_KEYS = ["calls", "input", "output", "cacheRead", "cacheWrite", "cacheWrite1h", "reasoning", "total", "cost", "unpriced"];
@@ -376,7 +459,10 @@ function rebuildUsage(u) {
376
459
  if (typeof u.piAi !== "string" || !/^\d+\.\d+\.\d+$/.test(u.piAi)) return null;
377
460
  piAi = u.piAi;
378
461
  }
379
- let truncated = 0;
462
+ // ABSENT stays null, never 0 (PR #549's review): a runner that did not report how many rows it folded did not
463
+ // measure it, and the per-model dollar settlement floors on anything but a present 0. Readers that count folded
464
+ // ledgers treat null as none (`?? 0`).
465
+ let truncated = null;
380
466
  if (u.truncated !== undefined) {
381
467
  if (!Number.isInteger(u.truncated) || u.truncated < 0) return null;
382
468
  truncated = u.truncated;
@@ -392,6 +478,7 @@ function rebuildUsage(u) {
392
478
  // itself never has to admit uppercase.
393
479
  const provider = row.provider.toLowerCase();
394
480
  const model = row.model.toLowerCase();
481
+ // The id allowlist (model-ref.mjs, widened for issues #501/#502 so every builtin catalog id passes).
395
482
  if (!USAGE_ID_PATTERN.test(provider) || !USAGE_ID_PATTERN.test(model)) return null;
396
483
  const nums = {};
397
484
  for (const key of USAGE_ROW_NUMERIC_KEYS) {
@@ -430,12 +517,12 @@ function rebuildUsage(u) {
430
517
  * path stay out of the record -- for local jobs only the folder's `basename` is kept, because the full
431
518
  * path embeds the operator's OS account name.
432
519
  *
433
- * `reason` is a fixed enum passthrough (worker-abort | over-budget | unprotected-branch |
520
+ * `reason` is a fixed enum passthrough (worker-abort | over-budget | dollar-cap | allocation-cap | envelope-mismatch | portfolio-no-envelope | portfolio-snapshot-oversize | unprotected-branch |
434
521
  * runner-policy | provider-auth-refused | job-image-missing | egress-proxy-missing | ...), never free-form or payload text. `exitCode`, `turns`, and `budgetReserved`
435
522
  * default to `null` when the outcome does not carry them, so the record shape is stable whether or not
436
523
  * the source reports those fields.
437
524
  */
438
- export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null }) {
525
+ export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null }) {
439
526
  const data = job.data ?? {};
440
527
  const kind = data.kind ?? job.name;
441
528
  const source = result ?? error ?? {};
@@ -545,9 +632,85 @@ export function buildRecord({ job, result, error, startedAt, endedAt, host = nul
545
632
  // default was passed (a dependency-injection seam; `recordRun` in start.mjs always passes one) or
546
633
  // where the job data carries an explicit null, which no producer writes and the blessed gate refuses.
547
634
  backend: resolveBackendName(data, defaultBackend),
635
+ // The dollar reservation's outcome (issue #501, INT-RUN-HISTORY-FILE-CONTRACT). Additive, nullable, an explicit
636
+ // literal REBUILT here (never the source's object), TAIL position after `backend` on the same contract: field order
637
+ // is the serialisation order. `{ reservedMicros, settledMicros, basis, modelBasis }`: two integers of micro-dollars,
638
+ // one fixed token (`metered` | `floor` | `refunded` | `unreserved`), and `modelBasis`, how the per-model windows
639
+ // settled (`metered` | `floor` | `refunded`, null when the job held none). Null when no dollar window applied to the job (no dollar setting, or a
640
+ // refusal before the reservation step), which is every record of a deployment that sets none.
641
+ dollars: dollarsOf(source.dollars),
642
+ // The refusal's detail (PR #558, the end-of-round check of #501 and #502): which rule of the reason refused the job, the same fixed token
643
+ // the worker's log line carries, e.g. `overlay-link` under `model-unknown`, so the drill-in can tell the
644
+ // deployment's file from the job's model. Additive, nullable, an explicit literal, TAIL position after `dollars`
645
+ // on the same contract. A token of the fixed charset (`WHY_RE`) or null, never a free string, so the record stays
646
+ // PII-free by construction; null for every reason that carries no detail.
647
+ why: typeof source.why === "string" && WHY_RE.test(source.why) ? source.why : null,
648
+ // The job's project (issue #499, INT-PROJECTS-FILE-CONTRACT). Additive, nullable, an explicit literal, TAIL position
649
+ // after `why` on the same contract. The project's ID only, never its `name`: the id is operator-authored and
650
+ // charset-checked (`PROJECT_ID_RE`), never payload, the argument `host` and `backend` make, so the record stays
651
+ // PII-free by construction; the name is free text and has no path here. Passed in, resolved at the pickup gate
652
+ // (start.mjs `recordRun`), so `buildRecord` stays pure. Anything that is not a well-formed id records null.
653
+ project: isProjectId(project) ? project : null,
654
+ // The priorities plan a completed portfolio job wrote (issue #505, INT-RUN-HISTORY-FILE-CONTRACT). Additive, nullable,
655
+ // an explicit literal REBUILT here, TAIL position after `project` on the same contract. `{ outcome, reason, planId,
656
+ // clamped }`: a fixed outcome, a fixed reason or null, a 16-hex content hash or null, a boolean. Null for every
657
+ // run that left no `/outbox/priorities.json` and was no confirmed portfolio job, which is every run of a deployment
658
+ // with no portfolio trigger; a confirmed one that wrote none says `plan-absent` (issue #507). Never
659
+ // the plan's weights or its reasons: those are in the allocation audit file, and a reason is agent text.
660
+ plan: planOf(source.plan),
548
661
  };
549
662
  }
550
663
 
664
+ /** What became of a collected plan. */
665
+ export const PLAN_RECORD_OUTCOMES = Object.freeze(["applied", "duplicate", "refused"]);
666
+ /**
667
+ * Every reason a record's `plan` may carry: the collector's own rungs (outbox-plan.mjs `PLAN_COLLECT_REASONS`), the
668
+ * plan's judgement (`plan-invalid`) and the apply ladder (priorities.mjs `PLAN_LADDER`, with `envelope-mismatch`).
669
+ * Restated here so this module imports neither, and pinned to them by a test.
670
+ */
671
+ export const PLAN_RECORD_REASONS = Object.freeze([
672
+ "plan-absent",
673
+ "plan-not-portfolio",
674
+ "plan-oversize",
675
+ "plan-not-regular-file",
676
+ "plan-unreadable",
677
+ "plan-parse-error",
678
+ "plan-collect-error",
679
+ "plan-invalid",
680
+ "delegation-off",
681
+ "writer-not-allowed",
682
+ "envelope-mismatch",
683
+ "plan-duplicate",
684
+ "plan-stale",
685
+ "plan-too-soon",
686
+ "plan-incomplete",
687
+ "plan-busy",
688
+ ]);
689
+ const PLAN_ID_RE = /^[0-9a-f]{16}$/;
690
+
691
+ /** The record's `plan`, rebuilt from named fields, or null when the source carries none or a malformed one. */
692
+ function planOf(p) {
693
+ if (p === null || typeof p !== "object" || !PLAN_RECORD_OUTCOMES.includes(p.outcome)) return null;
694
+ const reason = PLAN_RECORD_REASONS.includes(p.reason) ? p.reason : null;
695
+ // An applied plan has no reason, and a duplicate or a refusal always has one; a source that breaks that is not trusted.
696
+ if ((p.outcome === "applied") !== (reason === null)) return null;
697
+ return { outcome: p.outcome, reason, planId: typeof p.planId === "string" && PLAN_ID_RE.test(p.planId) ? p.planId : null, clamped: p.clamped === true };
698
+ }
699
+
700
+ /** The charset a record's `why` must match: a lowercase token, the shape of every `why` the processor returns. */
701
+ const WHY_RE = /^[a-z][a-z0-9-]{0,63}$/;
702
+
703
+ const DOLLAR_BASES = new Set(["metered", "floor", "refunded", "unreserved"]);
704
+ // How the job's MODEL windows settled (issue #502 part 6); anything else, null included, records null.
705
+ const MODEL_BASES = new Set(["metered", "floor", "refunded"]);
706
+
707
+ /** The record's `dollars`, rebuilt from named fields only, or null when the source carries none or a malformed one. */
708
+ function dollarsOf(d) {
709
+ if (d === null || typeof d !== "object") return null;
710
+ if (!Number.isSafeInteger(d.reservedMicros) || d.reservedMicros < 0 || !Number.isSafeInteger(d.settledMicros) || d.settledMicros < 0 || !DOLLAR_BASES.has(d.basis)) return null;
711
+ return { reservedMicros: d.reservedMicros, settledMicros: d.settledMicros, basis: d.basis, modelBasis: MODEL_BASES.has(d.modelBasis) ? d.modelBasis : null };
712
+ }
713
+
551
714
  /**
552
715
  * A stable, non-PII target label. Forge jobs read `repo<sep>number`; local jobs read `local:<basename>` --
553
716
  * basename only, so the full folder path (which on Windows carries the OS account name) never lands in
@@ -577,6 +740,49 @@ export function targetFor(kind, data) {
577
740
  /** Retain only the last ~8KB of container output for turn recovery, so per-job memory stays flat. */
578
741
  const TAIL_CAP_BYTES = 8 * 1024;
579
742
 
743
+ /** A signed exit line's tail: `,"auth":"<64 hex>"}`, the MAC as the LAST key (image/runner/src/exit-line.mjs). */
744
+ const SIGNED_TAIL = /,"auth":"([0-9a-f]{64})"\}$/;
745
+
746
+ /**
747
+ * The runner exit lines in `text` that carry a valid MAC under `key`, each with its `auth` key removed, joined by
748
+ * newlines in their original order; "" when there is none (issue #545, INT-RUNNER-EXIT-CODE-PROTOCOL).
749
+ *
750
+ * The worker hands each job a fresh random key on the container's stdin, and the runner signs its exit line with
751
+ * HMAC-SHA256 over the line's unsigned bytes. A job's own tool can write any bytes it likes to the container's stdout,
752
+ * but not that MAC: the key left the pipe before any tool existed and the runner's memory is closed to the job's uid
753
+ * (the image's exec-only node). So when a key was issued, only these lines are the runner's, and every `parseExit*`
754
+ * scanner runs over this text instead of the raw tail: a forged line, before or after the genuine one, is not in it.
755
+ *
756
+ * Each candidate starts at a `{"event":"exit"` anchor, the `parseTailLine` repair's reasoning: those raw bytes cannot
757
+ * occur inside a runner line, so a line glued to stray bytes before it is still found, and an anchor inside the stray
758
+ * bytes simply fails the MAC. The comparison is `timingSafeEqual` on equal-length digests. Never throws.
759
+ */
760
+ export function authenticExitLines(text, key) {
761
+ if (typeof text !== "string" || typeof key !== "string" || key === "") return "";
762
+ const verified = [];
763
+ for (const raw of text.split("\n")) {
764
+ const line = raw.trim();
765
+ const match = SIGNED_TAIL.exec(line);
766
+ if (match === null) continue;
767
+ const signed = line.slice(0, match.index);
768
+ let from = signed.indexOf('{"event":"exit"');
769
+ while (from !== -1) {
770
+ const body = `${signed.slice(from)}}`;
771
+ try {
772
+ const expected = createHmac("sha256", key).update(body, "utf8").digest();
773
+ if (timingSafeEqual(expected, Buffer.from(match[1], "hex"))) {
774
+ verified.push(body);
775
+ break;
776
+ }
777
+ } catch {
778
+ // a digest that will not compare is not a match
779
+ }
780
+ from = signed.indexOf('{"event":"exit"', from + 1);
781
+ }
782
+ }
783
+ return verified.join("\n");
784
+ }
785
+
580
786
  /**
581
787
  * The durable log sink: the I/O layer that streams a job's raw container output to a per-job `.log`
582
788
  * file and recovers the turn count from a bounded tail of that same output.
@@ -604,7 +810,10 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
604
810
  log("logs_dir_error", { reason: err?.message });
605
811
  }
606
812
 
607
- return function openJobLog(jobId) {
813
+ // `exitKey` (issue #545): the key this job's runner signs its exit line with, or null when none was issued (an image
814
+ // that does not declare `exitAuth`, or a caller that predates the field). With a key, the exit-line fields are read
815
+ // from the AUTHENTICATED lines only, and none at all reads exactly like a container that wrote no exit line.
816
+ return function openJobLog(jobId, { exitKey = null } = {}) {
608
817
  let tail = "";
609
818
  let stream = null;
610
819
 
@@ -630,12 +839,20 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
630
839
  // The tail accumulates whether or not `enabled` is set, which is what keeps exitReason in the record
631
840
  // when raw logs are off: a label that vanished with log capture would make the record depend on an
632
841
  // opt-in PII switch.
633
- const exitReason = parseExitReason(tail);
634
- const turns = parseExitTurns(tail);
635
- const tokens = parseExitTokens(tail);
636
- const session = parseExitSession(tail);
637
- const usage = parseExitUsage(tail);
638
- const context = parseExitContext(tail);
842
+ // With a key, the scanners see only the lines the runner signed (`authenticExitLines`), so a line a job's tool
843
+ // wrote, before or after the genuine one, is never the one they read. `exitAuth` says which case this was:
844
+ // absent (no key issued: today's whole-tail read), "verified" or "unverified" (a key, and no line carried it).
845
+ const keyed = typeof exitKey === "string" && exitKey !== "";
846
+ const exitText = keyed ? authenticExitLines(tail, exitKey) : tail;
847
+ const exitAuth = exitText === "" ? "unverified" : "verified";
848
+ const exitReason = parseExitReason(exitText);
849
+ const exitLineCode = parseExitCode(exitText);
850
+ const exitWhy = parseExitWhy(exitText);
851
+ const turns = parseExitTurns(exitText);
852
+ const tokens = parseExitTokens(exitText);
853
+ const session = parseExitSession(exitText);
854
+ const usage = parseExitUsage(exitText);
855
+ const context = parseExitContext(exitText);
639
856
  try {
640
857
  if (stream !== null) {
641
858
  const s = stream;
@@ -659,7 +876,9 @@ export function makeLogSink({ logsDir, enabled, fs = nodeFs, log = () => {} }) {
659
876
  } catch (err) {
660
877
  log("log_sink_error", { jobId, reason: err?.message });
661
878
  }
662
- return { turns, tokens, session, usage, context, exitReason };
879
+ // `exitAuth` only when a key was issued, so a keyless close returns exactly the object it always did. `exitWhy`
880
+ // (issue #507) only when the line named one, for the same reason.
881
+ return { turns, tokens, session, usage, context, exitReason, exitLineCode, ...(exitWhy !== null ? { exitWhy } : {}), ...(keyed ? { exitAuth } : {}) };
663
882
  }
664
883
 
665
884
  return { write, close };
@@ -716,6 +935,11 @@ export function makeRecordWriter({ logsDir, fs = nodeFs, log = () => {} }) {
716
935
  * (schedulers "a" vs "a_1": `repeat_a_1_100.json` has a non-digit tail after "repeat_a_", so it never
717
936
  * matches scheduler "a"). The max millis strictly below `beforeMillis` is the previous fire.
718
937
  *
938
+ * A fire by hand (`pi-dispatch run --trigger <id>`, issue #505) counts as a run of that trigger: its id is
939
+ * `manual:<schedulerId>:<millis>`, the millis floored to the minute it was queued, stored as `manual_<id>_<millis>.json`.
940
+ * It ran the trigger's own data, so the next tick's `previousRunAt` names it rather than an older scheduled fire. A hand
941
+ * fire and a tick in the same minute share a millis; then the one that ended later is the previous run.
942
+ *
719
943
  * Returns `record.endedAt ?? record.startedAt ?? null` as an ISO string. `endedAt` first: BullMQ never
720
944
  * overlaps two fires of one scheduler, so the prior run's end is the honest high-water mark; `startedAt`
721
945
  * covers a crashed run's partial record. ANY failure -- missing dir, no prior run, unreadable file, bad
@@ -729,26 +953,157 @@ export function makeFindPreviousRun({ logsDir, fs = nodeFs }) {
729
953
  return function findPreviousRun({ schedulerId, beforeMillis } = {}) {
730
954
  try {
731
955
  if (typeof beforeMillis !== "number" || !Number.isFinite(beforeMillis)) return null;
732
- const prefix = sanitizeJobId(`repeat:${schedulerId}:`);
733
- const escaped = prefix.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
734
- const pattern = new RegExp(`^${escaped}(\\d+)\\.json$`);
956
+ const escape = (text) => text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
957
+ const pattern = new RegExp(`^(?:${escape(sanitizeJobId(`repeat:${schedulerId}:`))}|${escape(sanitizeJobId(`manual:${schedulerId}:`))})(\\d+)\\.json$`);
735
958
  let best = null;
736
959
  for (const name of fs.readdirSync(logsDir)) {
737
960
  const m = pattern.exec(name);
738
961
  if (m === null) continue;
739
962
  const millis = Number(m[1]);
740
963
  if (!Number.isFinite(millis) || millis >= beforeMillis) continue;
741
- if (best === null || millis > best.millis) best = { millis, name };
964
+ if (best === null || millis > best.millis) best = { millis, names: [name] };
965
+ else if (millis === best.millis) best.names.push(name);
742
966
  }
743
967
  if (best === null) return null;
744
- const record = JSON.parse(fs.readFileSync(join(logsDir, best.name), "utf8"));
745
- return record.endedAt ?? record.startedAt ?? null;
968
+ let at = null;
969
+ for (const name of best.names) {
970
+ const record = JSON.parse(fs.readFileSync(join(logsDir, name), "utf8"));
971
+ const end = record.endedAt ?? record.startedAt ?? null;
972
+ if (typeof end === "string" && (at === null || end > at)) at = end;
973
+ }
974
+ return at;
746
975
  } catch {
747
976
  return null; // NEVER throws: a history fault must not fail the prepare that asked
748
977
  }
749
978
  };
750
979
  }
751
980
 
981
+ /**
982
+ * What `makeReadRecord` returns for a record file that exists but cannot be read as a record (unreadable, bad JSON,
983
+ * not an object). Distinct from `null` (no file), so the lost-lock check can say which it met.
984
+ */
985
+ export const UNREADABLE_RECORD = Object.freeze({ unreadable: true });
986
+
987
+ /**
988
+ * Read one job's run record back, by the name `makeRecordWriter` gave it: `<logsDir>/<sanitizeJobId(id)>.json`
989
+ * (INT-RUN-HISTORY-FILE-CONTRACT, unchanged). Returns the parsed object, `null` when there is no file, or
990
+ * `UNREADABLE_RECORD` when there is one that is not a record.
991
+ *
992
+ * Why the worker reads its own record: the queue can lose a job's lock after the processor finished and wrote
993
+ * this file, when Valkey was unreachable for longer than BullMQ's lock renewal window. BullMQ then refuses the
994
+ * completion ("Missing lock") and its stall check takes the job back, so the queue alone says the job failed or
995
+ * must run again. The record is the store that already knows it finished (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK).
996
+ *
997
+ * The id inside is NOT checked here: `sanitizeJobId` maps `a:b` and `a_b` to one name, and `recordVerdict` refuses
998
+ * the other job's record as `id-mismatch`, which is the one place that can also say so. NEVER throws.
999
+ */
1000
+ export function makeReadRecord({ logsDir, fs = nodeFs }) {
1001
+ return function readRecord(jobId) {
1002
+ if (jobId === null || jobId === undefined || jobId === "") return null;
1003
+ let text;
1004
+ try {
1005
+ text = fs.readFileSync(join(logsDir, `${sanitizeJobId(jobId)}.json`), "utf8");
1006
+ } catch (err) {
1007
+ return err?.code === "ENOENT" || err?.code === "ENOTDIR" ? null : UNREADABLE_RECORD;
1008
+ }
1009
+ try {
1010
+ const record = JSON.parse(text);
1011
+ return record !== null && typeof record === "object" && !Array.isArray(record) ? record : UNREADABLE_RECORD;
1012
+ } catch {
1013
+ return UNREADABLE_RECORD;
1014
+ }
1015
+ };
1016
+ }
1017
+
1018
+ /**
1019
+ * How far a record's `startedAt` may sit BEFORE the job's creation and still be this job's record: 5 minutes.
1020
+ *
1021
+ * The two instants come from two clocks. `job.timestamp` is stamped by whoever ENQUEUED the job (the receiver, the
1022
+ * CLI, another worker) and `startedAt` by the worker that ran it, so a producer whose clock runs ahead of the
1023
+ * worker's would make every genuine record look older than its job and silently turn this check off. The
1024
+ * tolerance absorbs ordinary skew between hosts; what it must still refuse is a record of an OLDER job under a
1025
+ * reused id, and such a record is older by at least that job's whole run, its removal from the queue and a new
1026
+ * enqueue, which in practice is far more than 5 minutes. A skew larger than this is an operator problem the
1027
+ * `job_lost_lock_record_rejected` line (`older-than-job`) makes visible.
1028
+ */
1029
+ export const RECORD_CLOCK_SKEW_MS = 5 * 60 * 1000;
1030
+
1031
+ /**
1032
+ * Does this record say that THIS attempt of THIS job finished without failing? Returns `null` when it does,
1033
+ * `"absent"` when there is no record, and otherwise the reason it does not, from a fixed set:
1034
+ * `unreadable`, `id-mismatch`, `other-attempt`, `failed`, `older-than-job`. Pure; never throws.
1035
+ *
1036
+ * `attempt` is the 1-based attempt number the record would carry, which is BullMQ's `attemptsMade + 1` while the
1037
+ * job is processing (see `buildRecord`) and `attemptsMade` after BullMQ failed it (its moveToFailed adds one
1038
+ * before the failed event). A record of an earlier attempt does not answer for a later one.
1039
+ *
1040
+ * `since` is the job's creation time in millis (BullMQ's `job.timestamp`). A job id can be used again once the
1041
+ * old job is gone (a fixed `--trigger` id, a redelivered webhook), while its record lives for the retention
1042
+ * window; a record that started before this job existed (less `RECORD_CLOCK_SKEW_MS`) belongs to the old one. A
1043
+ * missing or unparseable `startedAt` is not trusted for the same reason.
1044
+ *
1045
+ * `failed` (an outcome of `failed`, or none) is refused on purpose: a record written by the catch path says the run
1046
+ * threw, and the queue's verdict (failed, or run again) is then the right one.
1047
+ */
1048
+ export function recordVerdict(record, { jobId, attempt, since } = {}) {
1049
+ try {
1050
+ if (record === null || record === undefined) return "absent";
1051
+ if (record === UNREADABLE_RECORD || typeof record !== "object" || Array.isArray(record)) return "unreadable";
1052
+ if (record.jobId !== jobId) return "id-mismatch";
1053
+ if (!Number.isInteger(attempt) || record.attempt !== attempt) return "other-attempt";
1054
+ if (typeof record.outcome !== "string" || record.outcome === "failed") return "failed";
1055
+ if (Number.isFinite(since)) {
1056
+ const started = Date.parse(record.startedAt ?? "");
1057
+ if (!Number.isFinite(started) || started < since - RECORD_CLOCK_SKEW_MS) return "older-than-job";
1058
+ }
1059
+ return null;
1060
+ } catch {
1061
+ return "unreadable";
1062
+ }
1063
+ }
1064
+
1065
+ /** `recordVerdict` as a yes or no. */
1066
+ export const recordSettlesAttempt = (record, opts) => recordVerdict(record, opts) === null;
1067
+
1068
+ /**
1069
+ * The lookup the worker asks when the queue lost a job's lock: the local record first, then the fleet's copy.
1070
+ *
1071
+ * On a fleet without a shared `PI_LOGS_DIR`, the host that meets the stalled job is not always the host that ran
1072
+ * it, so the local file can be absent or hold an older attempt. The run mirror (`run-mirror.mjs`) holds the same
1073
+ * record's bytes, so it is asked when the local file does not settle the attempt. `readMirrored` is `null` on a
1074
+ * deployment that declared no worker name, which has no mirror and is one host.
1075
+ *
1076
+ * `onReject(reason, source)` is told about every record it FOUND and refused (`source` is `local` or `mirror`,
1077
+ * `reason` one of `recordVerdict`'s tokens, never `absent`), so the caller can log why the failure path stood.
1078
+ *
1079
+ * Resolves to the record or `null`, and NEVER rejects: a mirror fault is no record, which fails toward today's
1080
+ * behaviour (a failure comment, or a run).
1081
+ */
1082
+ export function makeSettledRecord({ readRecord, readMirrored = null }) {
1083
+ return async function settledRecord(jobId, { attempt, since, onReject = () => {} } = {}) {
1084
+ const judge = (record, source) => {
1085
+ const why = recordVerdict(record, { jobId, attempt, since });
1086
+ if (why !== null && why !== "absent") {
1087
+ try {
1088
+ onReject(why, source);
1089
+ } catch {
1090
+ // a reporting fault must not change the verdict
1091
+ }
1092
+ }
1093
+ return why === null;
1094
+ };
1095
+ try {
1096
+ const local = readRecord(jobId);
1097
+ if (judge(local, "local")) return local;
1098
+ if (typeof readMirrored !== "function") return null;
1099
+ const mirrored = await readMirrored(jobId);
1100
+ return judge(mirrored, "mirror") ? mirrored : null;
1101
+ } catch {
1102
+ return null;
1103
+ }
1104
+ };
1105
+ }
1106
+
752
1107
  /**
753
1108
  * The durable log reaper: an age sweep that deletes `.log` and `.json` history files older than the
754
1109
  * retention window, keeping the logs directory bounded. Runs at boot AND on the retention timer since
@@ -1,3 +1,5 @@
1
+ import { UNREADABLE_RECORD } from "./run-history.mjs";
2
+
1
3
  /**
2
4
  * A fleet-visible copy of the run history (issue #57, Gap 3).
3
5
  *
@@ -186,6 +188,34 @@ export async function readMirroredRuns(redis, { limit = 50, sinceMs = 0, now = (
186
188
  return { runs, degraded: runs.length >= limit ? "truncated" : "ok" };
187
189
  }
188
190
 
191
+ /**
192
+ * One run's mirrored record, by its sanitized id: the parsed object, `null`, or `UNREADABLE_RECORD`. Never throws,
193
+ * never rejects.
194
+ *
195
+ * The by-id read the lost-lock check needs (`makeSettledRecord` in `run-history.mjs`): one bounded `GET`, not the
196
+ * index scan `readMirroredRuns` does for the panel. Unreachable and absent are `null` (nothing found); a value that
197
+ * is there but is not a record (unparseable, empty, not an object) is `UNREADABLE_RECORD`, exactly as the local
198
+ * reader says it, so the check can log `unreadable` from `mirror`. Neither is ever accepted: the caller keeps
199
+ * today's path.
200
+ */
201
+ export async function readMirroredRecord(redis, sanitizedJobId, { timeoutMs = OP_TIMEOUT_MS } = {}) {
202
+ if (!redis || !sanitizedJobId) return null;
203
+ try {
204
+ const raw = await bounded(redis.get(runRecordKey(sanitizedJobId)), timeoutMs);
205
+ if (raw === null || raw === undefined) return null;
206
+ if (typeof raw !== "string" || raw === "") return UNREADABLE_RECORD;
207
+ let rec;
208
+ try {
209
+ rec = JSON.parse(raw);
210
+ } catch {
211
+ return UNREADABLE_RECORD;
212
+ }
213
+ return rec !== null && typeof rec === "object" && !Array.isArray(rec) ? rec : UNREADABLE_RECORD;
214
+ } catch {
215
+ return null; // unreachable or timed out: nothing was read
216
+ }
217
+ }
218
+
189
219
  /**
190
220
  * One list from two sources.
191
221
  *