@edgehero/pi-dispatch 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.7.0",
3
+ "version": "1.8.0",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -48,6 +48,7 @@
48
48
  "./open-browser": "./src/open-browser.mjs",
49
49
  "./git-dirty": "./src/git-dirty.mjs",
50
50
  "./queue": "./src/queue.mjs",
51
+ "./capabilities": "./src/capabilities.mjs",
51
52
  "./connection": "./src/connection.mjs",
52
53
  "./job-id": "./src/job-id.mjs",
53
54
  "./forges": "./src/forges.mjs",
@@ -66,6 +67,7 @@
66
67
  "./get-token": "./src/get-token.mjs",
67
68
  "./runtime-settings": "./src/runtime-settings.mjs",
68
69
  "./run-history": "./src/run-history.mjs",
70
+ "./run-mirror": "./src/run-mirror.mjs",
69
71
  "./sandbox": "./src/sandbox.mjs",
70
72
  "./sandbox-store": "./src/sandbox-store.mjs",
71
73
  "./subscriptions": "./src/subscriptions.mjs",
@@ -0,0 +1,179 @@
1
+ /**
2
+ * What a host can serve, and which queue a job that needs one of those things belongs on (issue #57,
3
+ * `OQ-032`).
4
+ *
5
+ * Two trigger fields bind a job to a MACHINE rather than to a repository: `run.secretsProfile` names a
6
+ * resolver the operator declared in that host's `PI_SECRET_PROFILES`, and a `run.waitFor` condition names a
7
+ * check script from its `PI_WAIT_PROFILES`. Both are refused pre-spend when the host popping the job has
8
+ * not declared them, and both refusals are RETURNED rather than thrown, so they are never retried. The
9
+ * same trigger therefore succeeds or fails depending on which worker happened to take the delivery, and it
10
+ * reads like a configuration error rather than a placement one.
11
+ *
12
+ * #57's Gap 2 exempted forge jobs on the grounds that their workspace is a fresh clone that any host can
13
+ * build. Issues #225 and #230 retracted that without saying so: a clone is portable, a resolver on one
14
+ * machine's disk is not.
15
+ *
16
+ * WHY THIS IS DECIDED AT ENQUEUE. The obvious alternative is to let any host take the job and defer it if
17
+ * it cannot serve it. That does not work here, and the reason is upstream rather than ours: BullMQ promotes
18
+ * a delayed job on EACH WORKER'S OWN CLOCK (`scripts.js` passes the client clock as the cut-off), so the
19
+ * host whose clock runs fastest wins every attempt, deterministically. If the host that cannot serve the
20
+ * job is the fast one, the job never reaches the one that can. Jitter changes when the attempt happens, not
21
+ * who wins it.
22
+ *
23
+ * THE ONE CAPABILITY DELIBERATELY NOT ROUTED is `run.resume`. A session key is `sha256(kind, repo, ref)`
24
+ * and `session-store.mjs` records that it is "not random... anyone who knows the repository and the branch
25
+ * can compute it", so publishing keys to route on them would disclose which repositories and branches a
26
+ * deployment works on, recoverable by guessing a repo name. That is more disclosing than everything else in
27
+ * the registry combined. A resume that lands on the wrong host cold-starts and says so in the record.
28
+ */
29
+
30
+ /** Class letters. Open enum: `g:` is reserved for forge credentials, the highest-value follow-on. */
31
+ export const CAP_SECRET = "s";
32
+ export const CAP_WAIT = "w";
33
+
34
+ /**
35
+ * The name charset, duplicated from `secret-profiles.mjs` and `wait-for.mjs` deliberately: this module
36
+ * imports nothing, and the two it copies already copy it from `triggers.mjs` for the same reason. What
37
+ * matters here is what the set EXCLUDES -- a comma, so the token list joins unambiguously, and a colon, so
38
+ * `<class>:<name>` decomposes at the first one.
39
+ */
40
+ const PROFILE_NAME = /^[A-Za-z0-9._-]+$/;
41
+
42
+ /**
43
+ * How fresh a host's registry row must be before a job is ROUTED to it.
44
+ *
45
+ * Not the 90s TTL, and the difference is the point. The TTL is a crash backstop: it answers "has this host
46
+ * definitely gone", and it is deliberately six missed beats so a blip cannot evict a working host from the
47
+ * panel. A routing decision needs the opposite polarity -- evidence of LIFE, not absence of expiry --
48
+ * because a job routed onto a dead host's queue sits there until that host comes back, and nothing else
49
+ * will take it. Three beats is late enough to ride out a slow beat and early enough that a stopped host
50
+ * stops attracting work long before its row expires.
51
+ */
52
+ export const ROUTE_FRESH_MS = 45_000;
53
+
54
+ /** One token, or null when the name is not one this deployment would accept. */
55
+ function token(cls, name) {
56
+ return typeof name === "string" && PROFILE_NAME.test(name) ? `${cls}:${name}` : null;
57
+ }
58
+
59
+ /**
60
+ * What THIS host can serve, from its own config, as a sorted token list.
61
+ *
62
+ * Sorted so the published string is stable: an unstable one would make the row differ every beat and any
63
+ * future fingerprint over it useless.
64
+ */
65
+ export function capabilityTokens({ secretProfiles, waitProfiles } = {}) {
66
+ const out = new Set();
67
+ for (const name of Object.keys(secretProfiles ?? {})) {
68
+ const t = token(CAP_SECRET, name);
69
+ if (t) out.add(t);
70
+ }
71
+ for (const name of Object.keys(waitProfiles ?? {})) {
72
+ const t = token(CAP_WAIT, name);
73
+ if (t) out.add(t);
74
+ }
75
+ return [...out].sort();
76
+ }
77
+
78
+ /** The registry value. Comma-joined, which the charset makes unambiguous. */
79
+ export function serializeCaps(tokens) {
80
+ return (tokens ?? []).join(",");
81
+ }
82
+
83
+ /**
84
+ * A peer's tokens, from its registry row. Peer-written, so every element is re-validated here rather than
85
+ * trusted: a row is written by another process and this one decides where money-spending work goes.
86
+ */
87
+ export function parseCaps(raw) {
88
+ const out = new Set();
89
+ for (const part of String(raw ?? "").split(",")) {
90
+ const at = part.indexOf(":");
91
+ if (at <= 0) continue;
92
+ const cls = part.slice(0, at);
93
+ const name = part.slice(at + 1);
94
+ if ((cls === CAP_SECRET || cls === CAP_WAIT) && PROFILE_NAME.test(name)) out.add(part);
95
+ }
96
+ return out;
97
+ }
98
+
99
+ /**
100
+ * What a JOB needs, from the same two fields the worker's own refusals read.
101
+ *
102
+ * `waitProfileNames`-shaped inline rather than imported, because this module is loaded by the RECEIVER and
103
+ * `wait-for.mjs` carries the whole wait engine. The extraction is three lines and the charset check below
104
+ * is the same one that module makes.
105
+ */
106
+ export function jobNeeds(job) {
107
+ const needs = new Set();
108
+ const secret = token(CAP_SECRET, job?.secretsProfile);
109
+ if (secret) needs.add(secret);
110
+ if (Array.isArray(job?.waitFor)) {
111
+ for (const condition of job.waitFor) {
112
+ const wait = token(CAP_WAIT, condition?.profile);
113
+ if (wait) needs.add(wait);
114
+ }
115
+ }
116
+ return [...needs].sort();
117
+ }
118
+
119
+ /**
120
+ * Which queue this job belongs on: a host queue name, or `null` for the shared queue.
121
+ *
122
+ * FOUR ABSTENTIONS, and each one lands on today's behaviour rather than on something new. That is what
123
+ * makes this safe to put in front of every forge delivery: the rule can only ever move a job that would
124
+ * otherwise have had a coin flip decide whether it ran.
125
+ *
126
+ * 1. The job needs nothing host-specific. The overwhelming majority of deliveries.
127
+ * 2. EVERY live host can serve it. The shared queue is then strictly better than picking one, because it
128
+ * load-balances, and it is what the docs tell operators to aim for by declaring the same profiles
129
+ * everywhere. A deployment that follows that advice is byte-identical to before.
130
+ * 3. NO host can serve it. Routing cannot help, and the shared queue produces the existing pre-spend
131
+ * refusal (`secret-profile-unknown` / `wait-profile-unknown`), which is the honest answer and already
132
+ * names what to fix. Inventing a new terminal state here would be worse than the one that exists.
133
+ * 4. No CAPABLE host has a queue of its own, or the registry could not be read. Nothing to route to.
134
+ *
135
+ * Otherwise the job goes to a capable host, chosen by hashing its jobId: deterministic, so a redelivery of
136
+ * the same job lands the same way and dedup still works, and spread, so a delivery fanned out into replicas
137
+ * does not pile every replica onto one machine.
138
+ */
139
+ export function routeForgeJob({ hosts, needs, jobId, now = () => Date.now(), freshMs = ROUTE_FRESH_MS } = {}) {
140
+ if (!Array.isArray(needs) || needs.length === 0) return null; // (1)
141
+ if (!Array.isArray(hosts) || hosts.length === 0) return null; // (4) unreadable registry, or nobody home
142
+
143
+ const live = hosts.filter((h) => {
144
+ // `staleMs` is derived by the reader; a row without one is a row we cannot date, and an undatable
145
+ // row is not evidence of life.
146
+ const stale = Number(h?.staleMs);
147
+ return Number.isFinite(stale) && stale <= freshMs;
148
+ });
149
+ if (live.length === 0) return null;
150
+
151
+ const serves = (h) => {
152
+ const caps = parseCaps(h?.caps);
153
+ return needs.every((n) => caps.has(n));
154
+ };
155
+ const capable = live.filter(serves);
156
+ if (capable.length === 0) return null; // (3)
157
+ if (capable.length === live.length) return null; // (2)
158
+
159
+ // Only a host that DECLARED a name drains a queue of its own; an undeclared one reads the shared queue
160
+ // only, so routing to it would be routing into a queue nothing drains.
161
+ const routable = capable.filter((h) => h?.routes === true || h?.routes === "true").map((h) => h?.name).filter((n) => typeof n === "string" && n !== "");
162
+ if (routable.length === 0) return null; // (4)
163
+
164
+ routable.sort();
165
+ return routable[hashIndex(String(jobId ?? ""), routable.length)];
166
+ }
167
+
168
+ /**
169
+ * A stable index from a string. FNV-1a, inline: this module imports nothing, and a cryptographic hash would
170
+ * be a strange dependency for choosing between two machines.
171
+ */
172
+ function hashIndex(text, modulo) {
173
+ let h = 0x811c9dc5;
174
+ for (let i = 0; i < text.length; i++) {
175
+ h ^= text.charCodeAt(i);
176
+ h = Math.imul(h, 0x01000193) >>> 0;
177
+ }
178
+ return h % modulo;
179
+ }
@@ -0,0 +1,221 @@
1
+ /**
2
+ * A fleet-visible copy of the run history (issue #57, Gap 3).
3
+ *
4
+ * Every worker writes its records to its own `PI_LOGS_DIR`, so on more than one machine each host's panel
5
+ * lists only the runs on its own disk. The operator sees a third of their deployment and has no way to
6
+ * know it.
7
+ *
8
+ * SHARED STORAGE IS THE OTHER ANSWER AND IT IS NOT SECOND-BEST. On a shared `PI_LOGS_DIR` the local read
9
+ * IS the merged read, with no machinery at all, and this module is redundant. It ships because that trade
10
+ * runs both ways and an operator must be allowed to decline it: sharing the directory also shares the
11
+ * PII-bearing raw `.log`, and a mount outage becomes a LOST RECORD where a Valkey outage costs only a
12
+ * fleet view. Both shapes work; `docs/multi-host.md` says which is which.
13
+ *
14
+ * THE FILE IS THE RECORD AND THIS IS A VIEW. Three things follow, and each is load-bearing:
15
+ *
16
+ * - The file is written FIRST, always. A crash between the two leaves a fleet-visible run whose durable
17
+ * source does not exist, which inverts the one claim this design rests on.
18
+ * - The mirror's TTL is never longer than the file retention window, so it can never show a run whose
19
+ * file has already been reaped. A view that outlives its source is a second source of truth, which is
20
+ * exactly what `DES-RUN-HISTORY-FLAT-FILES-NO-DB` refuses.
21
+ * - Nothing derived is stored. The bytes are the sidecar's own bytes, so there is nothing to be stale
22
+ * RELATIVE TO: a retry overwrites the same key exactly as it overwrites the same file, and cost
23
+ * classification is still computed at fold time from `subscriptions.json` rather than frozen here.
24
+ *
25
+ * WHY THE WHOLE RECORD RATHER THAN A PROJECTION. The record is PII-free BY CONSTRUCTION -- it holds no
26
+ * attacker-chosen string, which `INT-RUN-HISTORY-FILE-CONTRACT` states and `buildRecord` enforces field by
27
+ * field. Copying it whole inherits that property; a projection would re-derive it at a second serialiser,
28
+ * where the next person to add a field has to remember this file exists. It is also what the readers need:
29
+ * the cost fold, the graph and the insights view read eleven fields between them.
30
+ *
31
+ * NOT MIRRORED: the raw `.log`. It is the one artifact here that holds issue text, comment text and tool
32
+ * output, and mirroring it would move that off the machine the operator chose to keep it on. A foreign
33
+ * run's record names its host, so the panel can say where the bytes are rather than pretending there are
34
+ * none.
35
+ */
36
+
37
+ /** The index: sanitized jobId -> the run's end (or start) in millis. */
38
+ export const RUNS_INDEX = "runs:index";
39
+
40
+ /** One run's own bytes. */
41
+ export const runRecordKey = (sanitizedJobId) => `runs:rec:${sanitizedJobId}`;
42
+
43
+ /**
44
+ * The deepest any reader asks. `SCAN_WINDOW_MAX_DAYS` in the admin is 92, so a longer window would hold
45
+ * bytes nothing can request.
46
+ */
47
+ export const MIRROR_MAX_DAYS = 92;
48
+
49
+ /**
50
+ * A hard ceiling on index members, independent of the time window.
51
+ *
52
+ * The window alone does not bound memory: a deployment running thousands of jobs a day would hold a
53
+ * quarter of a million members for ninety-two days. This caps what the fleet view can cost at roughly the
54
+ * depth a panel can display, and the file on disk remains the complete history either way.
55
+ */
56
+ export const RUNS_INDEX_MAX = 5_000;
57
+
58
+ const DAY_MS = 24 * 60 * 60 * 1000;
59
+ const OP_TIMEOUT_MS = 2_000;
60
+
61
+ /**
62
+ * How long a mirrored record lives.
63
+ *
64
+ * Never longer than the operator's own retention, and never longer than what any reader asks for.
65
+ * `retentionDays: 0` means keep the files forever, which is the one case where the mirror is the shorter
66
+ * of the two, so it clamps to the reader's ceiling rather than to infinity.
67
+ */
68
+ export function mirrorWindowMs(retentionDays) {
69
+ const days = Number(retentionDays) > 0 ? Math.min(Number(retentionDays), MIRROR_MAX_DAYS) : MIRROR_MAX_DAYS;
70
+ return days * DAY_MS;
71
+ }
72
+
73
+ /**
74
+ * BullMQ's connections carry `maxRetriesPerRequest: null`, so a command against an unreachable server
75
+ * QUEUES FOREVER rather than rejecting. Every await here is bounded for that reason; an unbounded one
76
+ * would not fail the mirror, it would hang the job that was writing to it.
77
+ */
78
+ function bounded(promise, ms) {
79
+ let timer;
80
+ return Promise.race([
81
+ promise,
82
+ new Promise((_, reject) => {
83
+ timer = setTimeout(() => reject(new Error("mirror timeout")), ms);
84
+ }),
85
+ ]).finally(() => clearTimeout(timer));
86
+ }
87
+
88
+ /**
89
+ * The writer. Returns `{ mirror, close }`; `mirror` never throws and never rejects.
90
+ *
91
+ * A history blip must not fail a job that has already run and already been recorded to disk. Every failure
92
+ * here costs a row in a fleet view and nothing else, which is why the whole body is wrapped and the result
93
+ * is a boolean nobody is obliged to read.
94
+ */
95
+ export function makeRunMirror({ redis, retentionDays, now = () => Date.now(), log = () => {}, timeoutMs = OP_TIMEOUT_MS, indexMax = RUNS_INDEX_MAX } = {}) {
96
+ const windowMs = mirrorWindowMs(retentionDays);
97
+ let warned = false;
98
+
99
+ return {
100
+ async mirror(record, sanitizedJobId) {
101
+ if (!redis || !record || !sanitizedJobId) return false;
102
+ try {
103
+ const at = Date.parse(record.endedAt ?? record.startedAt ?? "");
104
+ const score = Number.isFinite(at) ? at : now();
105
+ const body = JSON.stringify(record);
106
+ await bounded(redis.set(runRecordKey(sanitizedJobId), body, "PX", windowMs), timeoutMs);
107
+ await bounded(redis.zadd(RUNS_INDEX, score, sanitizedJobId), timeoutMs);
108
+ // Trimmed by the WRITER, twice: by age, and by count. Two `ZREMRANGE`s against a run that
109
+ // took minutes is free, and it means no reader has to pay for a backlog it did not create.
110
+ await bounded(redis.zremrangebyscore(RUNS_INDEX, "-inf", `(${now() - windowMs}`), timeoutMs);
111
+ await bounded(redis.zremrangebyrank(RUNS_INDEX, 0, -indexMax - 1), timeoutMs);
112
+ // ROLLING expiry, deliberately unlike `budget.mjs`'s set-once rule and deliberately like
113
+ // `pi-dispatch:sched-stalls`. A budget window must not be pushed forward by traffic or a busy
114
+ // day never resets; an ACTIVITY index should roll with traffic, because that is what it
115
+ // describes. A fleet that stops running jobs loses its index one window later, which is
116
+ // correct: there is nothing left to show.
117
+ await bounded(redis.pexpire(RUNS_INDEX, windowMs), timeoutMs);
118
+ warned = false;
119
+ return true;
120
+ } catch (err) {
121
+ // Once per transition, not once per job: a Valkey outage during a busy hour must not turn one
122
+ // fault into a thousand log lines (`notePackageKey`'s precedent).
123
+ if (!warned) {
124
+ warned = true;
125
+ log("run_mirror_failed", { jobId: sanitizedJobId, reason: err?.message });
126
+ }
127
+ return false;
128
+ }
129
+ },
130
+ };
131
+ }
132
+
133
+ /**
134
+ * The reader. Returns `{ runs, degraded }` and never throws.
135
+ *
136
+ * `degraded` is a DISCRIMINATED channel rather than a silence, and two of its values must not collapse:
137
+ * `"off"` means the index is absent, which is what a single-host deployment and a fleet of workers still
138
+ * below the version floor both look like, while `"unreachable"` means we could not tell. A new panel
139
+ * meeting old workers has to read "off", not "error".
140
+ *
141
+ * Two round trips regardless of how many runs come back: one `ZREVRANGEBYSCORE` for the ids, one `MGET`
142
+ * for the bodies. Never one read per run.
143
+ */
144
+ export async function readMirroredRuns(redis, { limit = 50, sinceMs = 0, now = () => Date.now(), timeoutMs = OP_TIMEOUT_MS } = {}) {
145
+ if (!redis) return { runs: [], degraded: "off" };
146
+ let ids;
147
+ try {
148
+ ids = await bounded(redis.zrevrangebyscore(RUNS_INDEX, "+inf", `(${sinceMs}`, "LIMIT", 0, Math.max(1, limit)), timeoutMs);
149
+ } catch (err) {
150
+ return { runs: [], degraded: `unreachable (${err?.message ?? "?"})` };
151
+ }
152
+ if (!Array.isArray(ids) || ids.length === 0) return { runs: [], degraded: "off" };
153
+
154
+ let bodies;
155
+ try {
156
+ bodies = await bounded(redis.mget(...ids.map(runRecordKey)), timeoutMs);
157
+ } catch (err) {
158
+ return { runs: [], degraded: `unreachable (${err?.message ?? "?"})` };
159
+ }
160
+
161
+ const runs = [];
162
+ const stale = [];
163
+ for (let i = 0; i < ids.length; i++) {
164
+ const raw = bodies?.[i];
165
+ if (typeof raw !== "string" || raw === "") {
166
+ // An id whose body has expired: the per-key TTL fired and the index member outlived it. The
167
+ // READER prunes it, which is what `wait:held` does for the same shape and for the same reason --
168
+ // a writer that crashed cannot clean up after itself, and a reader is already here.
169
+ stale.push(ids[i]);
170
+ continue;
171
+ }
172
+ try {
173
+ const rec = JSON.parse(raw);
174
+ if (rec && typeof rec === "object") runs.push(rec);
175
+ } catch {
176
+ stale.push(ids[i]); // unparseable is indistinguishable from gone, and equally not showable
177
+ }
178
+ }
179
+ if (stale.length > 0) {
180
+ try {
181
+ await bounded(redis.zrem(RUNS_INDEX, ...stale), timeoutMs);
182
+ } catch {
183
+ // best-effort: a straggler in the index costs one skipped row next time, never a wrong one
184
+ }
185
+ }
186
+ return { runs, degraded: runs.length >= limit ? "truncated" : "ok" };
187
+ }
188
+
189
+ /**
190
+ * One list from two sources.
191
+ *
192
+ * DEDUP BY LATER `endedAt`, LOCAL BREAKS A TIE. Not decoration: a retry can land on a different host, so
193
+ * host A may hold attempt 0 (failed) while host B mirrored attempt 1 (completed). "Local wins" alone would
194
+ * show the stale one. Local breaking an exact tie keeps a single-host deployment reading its own files.
195
+ *
196
+ * CUT AFTER THE SORT, never before. Slicing first is the defect `held.test.mjs` already exists to prevent:
197
+ * it makes the result depend on which source happened to be longer.
198
+ */
199
+ export function mergeRuns(local, mirrored, { limit = 50 } = {}) {
200
+ const by = new Map();
201
+ const at = (r) => {
202
+ const t = Date.parse(r?.endedAt ?? r?.startedAt ?? "");
203
+ return Number.isFinite(t) ? t : -Infinity;
204
+ };
205
+ // Mirrored first, so a local record with an equal timestamp overwrites it on the second pass.
206
+ for (const r of Array.isArray(mirrored) ? mirrored : []) if (r?.jobId) by.set(r.jobId, r);
207
+ for (const r of Array.isArray(local) ? local : []) {
208
+ if (!r?.jobId) continue;
209
+ const seen = by.get(r.jobId);
210
+ if (!seen || at(r) >= at(seen)) by.set(r.jobId, r);
211
+ }
212
+ const out = [...by.values()].sort((a, b) => at(b) - at(a));
213
+ return out.slice(0, Math.max(0, limit));
214
+ }
215
+
216
+ /** The distinct hosts a merged list came from, computed from the RECORDS rather than from the mirror. */
217
+ export function hostsIn(runs) {
218
+ const names = new Set();
219
+ for (const r of runs ?? []) if (typeof r?.host === "string" && r.host !== "") names.add(r.host);
220
+ return [...names].sort();
221
+ }
package/src/start.mjs CHANGED
@@ -15,6 +15,7 @@ import { makeAzureAuth } from "./azure-auth.mjs";
15
15
  import { makeAzureHost } from "./azure-host.mjs";
16
16
  import { makeEgressPreflight } from "./egress.mjs";
17
17
  import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
18
+ import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
18
19
  import { cronFingerprint } from "./fingerprint.mjs";
19
20
  import { makeHostRegistry } from "./host-registry.mjs";
20
21
  import { makeImagePreflight } from "./image-preflight.mjs";
@@ -33,7 +34,8 @@ import { makeWaitState } from "./wait-state.mjs";
33
34
  import { hostQueueName, makeQueue } from "./queue.mjs";
34
35
  import { makeRunContainer } from "./run-container.mjs";
35
36
  import { makeSecretsResolver } from "./secrets.mjs";
36
- import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter } from "./run-history.mjs";
37
+ import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, sanitizeJobId } from "./run-history.mjs";
38
+ import { makeRunMirror } from "./run-mirror.mjs";
37
39
  import { effectiveSettings, readOverlay } from "./runtime-settings.mjs";
38
40
  import { authoredCron, loadSchedules, servedSchedules } from "./schedules.mjs";
39
41
  import { makeStallGuard } from "./scheduler-stall-guard.mjs";
@@ -237,6 +239,7 @@ export async function startWorker(
237
239
  makeReaper: makeReaperFn = makeReaper,
238
240
  makeLogSink: makeLogSinkFn = makeLogSink,
239
241
  makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
242
+ makeRunMirror: makeRunMirrorFn = makeRunMirror,
240
243
  makeLogReaper: makeLogReaperFn = makeLogReaper,
241
244
  makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
242
245
  makeRunContainer: makeRunContainerFn = makeRunContainer,
@@ -446,10 +449,22 @@ export async function startWorker(
446
449
  // that can neither disarm nor pre-spend-check.
447
450
  const onceTriggersFile = env.PI_TRIGGERS_FILE ?? join(process.cwd(), "triggers.json");
448
451
  const disarmOnce = makeDisarmOnce({ triggersPath: onceTriggersFile, log });
452
+ // The fleet-visible copy of the run history (issue #57, Gap 3). Armed only on a deployment that declared
453
+ // a worker name: an unnamed one is a single host, its own files ARE the whole history, and a mirror
454
+ // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
455
+ // no job then issues a single extra Valkey command.
456
+ const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
449
457
  const recordRun = ({ job, result, error, startedAt, endedAt }) => {
450
458
  // The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
451
459
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
452
- writeRecord(buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName }));
460
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName });
461
+ writeRecord(record);
462
+ // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
463
+ // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
464
+ // and a view that can outlive its source is a second source of truth. Not awaited, because this is
465
+ // the job's own completion path: a slow Valkey may cost a row in a panel and must never hold up a
466
+ // job that has already finished and already been written to disk. `mirror` never rejects.
467
+ void runMirror?.mirror(record, sanitizeJobId(record.jobId));
453
468
  // Strictly AFTER the durable record: "fired" means "produced a run record", and the crash
454
469
  // direction this ordering buys is the chosen one -- an armed one-shot with a record, never a
455
470
  // disarm before writeRecord RETURNED. Returned, not succeeded: the record writer swallows fs
@@ -585,6 +600,13 @@ export async function startWorker(
585
600
  // Whether this host DRAINS a queue of its own. Every worker publishes a row; only a host that declared
586
601
  // a name has somewhere for routed work to go, and a reader must not invent a queue for one that has not.
587
602
  routes: config.workerNameDeclared,
603
+ // What this host can serve that another might not (issue #57, `OQ-032`): the secret and wait profiles
604
+ // it has declared. NAMES only, never the resolver paths behind them -- a path is PII on Windows and
605
+ // operator topology everywhere, and the receiver only needs to know WHICH host, not what it runs.
606
+ //
607
+ // Recomputed per beat rather than frozen at boot, for the reason the digest is: a host that gains a
608
+ // profile on restart must start attracting that work within one beat, and one that loses it must stop.
609
+ caps: () => serializeCaps(capabilityTokens(config)),
588
610
  // The host's IANA zone, because a cron PATTERN carries none: `triggers.json` has no `tz` field and
589
611
  // BullMQ hands the pattern to cron-parser with no zone, so it resolves in each worker's LOCAL time.
590
612
  // On one host that is exactly what an operator means; on two in different zones the same pattern is