@edgehero/pi-dispatch 1.3.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,403 @@
1
+ /**
2
+ * The shared triggers-file WRITER (issue #231, DES-ONE-SHOT-DISARM-IN-THE-FILE, OQ-008).
3
+ *
4
+ * Moved here from the admin's read-model so BOTH authors of `triggers.json` serialize through one
5
+ * funnel: the operator's console (dialogs and confirm-gated tools, via the admin's re-export) and the
6
+ * worker's one-shot disarm. "REUSE, NEVER RE-DERIVE" -- the same rule that single-sources the parser
7
+ * and `loadGitHubAuth`. The file format is the admin's own: 2-space JSON plus a trailing newline,
8
+ * byte-for-byte what `writeTriggers` always wrote, and a test pins it.
9
+ *
10
+ * THE LOCK. Until #231 there was exactly one writer (one single-threaded pi process, tools declared
11
+ * sequential), so read-modify-write had nothing to race and `renameSync`'s last-writer-wins was moot.
12
+ * The worker's disarm is a second author -- and at PI_CONCURRENCY up to 3, a third and fourth -- so
13
+ * every write now takes `<path>.lock` via exclusive create (`wx`), the session-store's idiom with its
14
+ * two doctrines kept verbatim: EEXIST is the ONLY failure that means locked (anything else failed to
15
+ * create the lock for its own reason and is reported as that reason), and a leaked lock is logged,
16
+ * never thrown. What is NEW here, with no in-repo precedent, is the STALE TAKEOVER: a lock whose
17
+ * mtime is older than LOCK_STALE_MS is unlinked and retaken once. The session store can afford to
18
+ * discard on contention and let its reaper sweep a leak; this file cannot -- a crashed writer's lock
19
+ * would otherwise wedge every trigger add, edit, delete and disarm on the deployment forever, and
20
+ * there is no reaper whose beat covers it. The residual is the classic one: unlink-then-create is not
21
+ * atomic, so two writers racing a stale takeover can interleave in a window of milliseconds. That
22
+ * window replaces today's always-open one, and the loser's write still validated through the shared
23
+ * parser, so the file stays loadable; the lost update is one disarm or one edit, and the disarm
24
+ * caller retries.
25
+ *
26
+ * Callers split by posture, deliberately:
27
+ * - `writeTriggers` is SYNC and gives up IMMEDIATELY on contention (`{ invalid }` naming the lock).
28
+ * Its callers sit on the pi TUI's event loop, where a bounded-retry sleep is a frozen panel; an
29
+ * operator whose keypress lost the race gets a message and presses the key again.
30
+ * - `disarmTrigger` is ASYNC and retries with jitter, because ITS caller is the worker's post-run
31
+ * hook with nobody at the keyboard, and the thing it races (an operator edit, a sibling job's
32
+ * disarm) clears in milliseconds.
33
+ *
34
+ * `disarmTrigger` is also deliberately NARROWER than `writeTriggers`: it refuses an unreadable file
35
+ * outright rather than repairing from empty. The repair posture is right for the operator CRUD path
36
+ * (a missing file plus "add trigger" should scaffold) and catastrophic here -- overwriting a file the
37
+ * worker could not read, to record one disarm, would destroy the operator's trigger set.
38
+ *
39
+ * Custom: exclusive-create lockfile per session-store.mjs precedent; no proper-lockfile dependency
40
+ * (repo keeps runtime deps minimal, and the two-writer case needs no lease/renewal machinery)
41
+ */
42
+
43
+ import nodeFs from "node:fs";
44
+ import { parseTriggers } from "./triggers.mjs";
45
+
46
+ /**
47
+ * A lock older than this is a crashed writer's, not a live one's: every write under it is a read,
48
+ * one mutate, one serialize and two syscalls, three orders of magnitude faster. Ten seconds rather
49
+ * than one so a laptop suspending mid-write on battery does not get its live lock stolen on resume.
50
+ */
51
+ const LOCK_STALE_MS = 10_000;
52
+
53
+ /** Bounded contention retry for the disarm path: ~5 attempts x 100-300ms jitter, well under a second of
54
+ * real contention, and the whole wait is smaller than the dedup window that bounds what a lost disarm
55
+ * costs. */
56
+ const DISARM_LOCK_ATTEMPTS = 5;
57
+
58
+ function lockPathFor(triggersPath) {
59
+ return `${triggersPath}.lock`;
60
+ }
61
+
62
+ /**
63
+ * Take the lock, with one stale takeover. Returns an fd, or null when a LIVE writer holds it.
64
+ * Throws only for non-EEXIST failures -- the session-store doctrine: reporting a read-only dir or a
65
+ * full disk as "locked" sends an operator hunting for a stuck lock file that does not exist.
66
+ */
67
+ function takeLock(triggersPath, fs, log) {
68
+ const lock = lockPathFor(triggersPath);
69
+ let sweptAgeMs = null;
70
+ for (let attempt = 0; attempt < 2; attempt++) {
71
+ try {
72
+ const fd = fs.openSync(lock, "wx"); // exclusive create IS the lock; no daemon, no lease
73
+ // Logged only AFTER the retake create succeeded: the unlink alone proves nothing (a rival
74
+ // sweeper may win the recreate race), and a takeover log for a lock we did not get would
75
+ // send an operator reading a history that never happened.
76
+ if (sweptAgeMs !== null) log("triggers_lock_stale_taken", { ageMs: sweptAgeMs });
77
+ return fd;
78
+ } catch (err) {
79
+ if (err?.code !== "EEXIST") throw err;
80
+ let mtimeMs;
81
+ try {
82
+ mtimeMs = fs.statSync(lock).mtimeMs;
83
+ } catch {
84
+ // The holder released between our open and our stat: the next loop iteration takes it.
85
+ continue;
86
+ }
87
+ if (Date.now() - mtimeMs <= LOCK_STALE_MS) return null; // live writer; caller decides
88
+ try {
89
+ fs.unlinkSync(lock);
90
+ } catch {
91
+ // Someone else swept it first; the retry create answers who won.
92
+ }
93
+ sweptAgeMs = Math.round(Date.now() - mtimeMs);
94
+ }
95
+ }
96
+ return null;
97
+ }
98
+
99
+ function releaseLock(fd, triggersPath, fs, log) {
100
+ fs.closeSync(fd);
101
+ try {
102
+ fs.unlinkSync(lockPathFor(triggersPath));
103
+ } catch {
104
+ // A leaked lock delays writers by LOCK_STALE_MS, then the takeover clears it. Logged, never
105
+ // thrown -- the session-store rule.
106
+ log("triggers_lock_stuck", {});
107
+ }
108
+ }
109
+
110
+ /** The one serializer: 2-space plus trailing newline, byte-identical to what the admin always wrote --
111
+ * both live-reload watchers and the operator's own diff read this file, so its shape is a contract. */
112
+ function serialize(triggersArray) {
113
+ return `${JSON.stringify({ triggers: triggersArray }, null, 2)}\n`;
114
+ }
115
+
116
+ /** tmp + rename with writeOverlay's single EPERM retry (a Windows AV/indexer briefly holding the
117
+ * destination); a second EPERM, and every other fs failure, propagates to the caller's contract. */
118
+ function renameIntoPlace(fs, tmp, dest) {
119
+ try {
120
+ fs.renameSync(tmp, dest);
121
+ } catch (err) {
122
+ if (err?.code !== "EPERM") throw err;
123
+ fs.renameSync(tmp, dest); // single retry: the AV/indexer lock is transient
124
+ }
125
+ }
126
+
127
+ let tmpSeq = 0;
128
+
129
+ /**
130
+ * A PER-WRITER tmp name, not the fixed `.tmp` the admin writer used to share. With one writer the
131
+ * fixed name was self-cleaning and harmless; with two authors it quietly voided the atomicity claim
132
+ * in the one window the lock concedes (the stale-takeover double-take): two writers sharing one tmp
133
+ * path means B can rename A's half-flushed tmp over the destination, and "a watcher never observes a
134
+ * half-written file" stops being true precisely when it matters. A pid+sequence name gives each
135
+ * write its own inode, so the rename is atomic no matter who else is mid-write. The cost is that a
136
+ * crash between write and rename leaves a uniquely-named straggler instead of one that the next
137
+ * write overwrites -- so both writers unlink their tmp on the failure path.
138
+ */
139
+ function tmpPathFor(triggersPath) {
140
+ return `${triggersPath}.${process.pid}.${tmpSeq++}.tmp`;
141
+ }
142
+
143
+ /** Best-effort cleanup of this writer's own tmp after a failed write; the file either renamed away
144
+ * (unlink finds nothing, fine) or must not be left as litter. Never throws over the real failure. */
145
+ function discardTmp(fs, tmp) {
146
+ try {
147
+ fs.unlinkSync(tmp);
148
+ } catch {
149
+ // Already renamed, or never written: either way there is nothing to clean.
150
+ }
151
+ }
152
+
153
+ /**
154
+ * Read-modify-write the triggers file under the lock. `mutate` receives the RAW entries (shallow
155
+ * copies) and returns the next raw array; the result is validated through the loaders' own
156
+ * `parseTriggers` -- never write a file they would reject -- and written atomically (tmp + rename) so
157
+ * a live-reload watcher never observes a half-written file.
158
+ *
159
+ * Human-approved writes only on the console path: reached from the operator-typed `/dispatch trigger
160
+ * ...` handlers and from the `dispatch_trigger_*` tools behind `confirmedWrite`'s dialog, so the
161
+ * human keypress is the approval and CONST-TRIGGER-AUTHOR-GATE's principle holds. The worker's disarm
162
+ * does NOT come through here -- `disarmTrigger` below is its own, narrower entry.
163
+ *
164
+ * A missing or unparseable existing file starts from an empty set; the validated write repairs it.
165
+ * Returns `{ ok: true }`, or `{ invalid }` for a validation failure OR a held lock; fs failures throw,
166
+ * the contract this function has always had.
167
+ */
168
+ export function writeTriggers({ triggersPath, mutate, fs = nodeFs, log = () => {} }) {
169
+ const fd = takeLock(triggersPath, fs, log);
170
+ if (fd === null) {
171
+ // Immediate, not retried: the callers sit on the pi TUI event loop, and the holder is a write
172
+ // that finishes in milliseconds. The operator re-presses; the file was never touched.
173
+ return { invalid: `triggers file locked (another write in progress): ${lockPathFor(triggersPath)}` };
174
+ }
175
+ try {
176
+ let current = [];
177
+ try {
178
+ const raw = JSON.parse(fs.readFileSync(triggersPath, "utf8"));
179
+ if (Array.isArray(raw?.triggers)) current = raw.triggers;
180
+ } catch {
181
+ // Missing/invalid file: start from empty; the validated atomic write below repairs it.
182
+ }
183
+ const next = mutate(current.map((t) => ({ ...t })));
184
+ const text = serialize(next);
185
+ try {
186
+ parseTriggers(text, triggersPath); // the loaders' own validator -- never write a file they would reject
187
+ } catch (e) {
188
+ return { invalid: e?.message ?? String(e) };
189
+ }
190
+ const tmp = tmpPathFor(triggersPath);
191
+ try {
192
+ fs.writeFileSync(tmp, text, { mode: 0o644 });
193
+ renameIntoPlace(fs, tmp, triggersPath);
194
+ } catch (err) {
195
+ discardTmp(fs, tmp);
196
+ throw err; // the writer's contract: fs failures throw
197
+ }
198
+ return { ok: true };
199
+ } finally {
200
+ releaseLock(fd, triggersPath, fs, log);
201
+ }
202
+ }
203
+
204
+ const sleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));
205
+
206
+ /**
207
+ * Disarm ONE spent one-shot: add `on.disarmed = { at, jobId }` to the entry at `index`, and nothing
208
+ * else -- the worker's whole write authority over this file is this one added key, which is what lets
209
+ * OQ-008's "the file is the single write target" survive a second author (the worker can disarm what
210
+ * an operator armed; no machine path can ARM anything).
211
+ *
212
+ * The identity check compares EVERY field the job knows about the trigger it matched: `index` is
213
+ * positional and the file can change between enqueue and disarm, so the entry must still be an armed
214
+ * one-shot naming the exact item (`number`) and dispatching the exact `flow` or `command` the job
215
+ * carried. What the check confirms is therefore the matched ITEM and TARGET, not the trigger
216
+ * INSTANCE: an operator who deletes a spent-in-flight one-shot and re-arms an IDENTICAL one (same
217
+ * index, same number, same flow) inside the job's own run window has re-armed something this writer
218
+ * cannot tell from the original, and the earlier job's disarm will spend it. That residual is
219
+ * named rather than closed because a per-trigger id is the thing the design rejects -- the raw index
220
+ * IS the identity (`INT-TRIGGERS-FILE-CONTRACT`) -- and every DIFFERING re-arm (other flow, other
221
+ * number, other shape) refuses loudly here.
222
+ *
223
+ * Returns `{ ok }`, `{ already }` (a sibling replica/redelivery won the race -- idempotent success,
224
+ * not failure), or `{ invalid }` with an operator-actionable reason. NEVER throws, and NEVER repairs:
225
+ * an unreadable file is `{ invalid }` with the bytes untouched.
226
+ */
227
+ export async function disarmTrigger({ triggersPath, index, number, flow, command, jobId, at, fs = nodeFs, log = () => {} }) {
228
+ for (let attempt = 0; attempt < DISARM_LOCK_ATTEMPTS; attempt++) {
229
+ let fd;
230
+ try {
231
+ fd = takeLock(triggersPath, fs, log);
232
+ } catch (err) {
233
+ return { invalid: `triggers file lock failed (${err?.code ?? "lock-error"}): ${triggersPath}` };
234
+ }
235
+ if (fd === null) {
236
+ await sleep(100 + Math.floor(Math.random() * 200));
237
+ continue;
238
+ }
239
+ try {
240
+ let raw;
241
+ try {
242
+ raw = JSON.parse(fs.readFileSync(triggersPath, "utf8"));
243
+ } catch (err) {
244
+ // NEVER repair-from-empty here: overwriting a file we could not read, to record one
245
+ // disarm, would destroy the operator's trigger set. writeTriggers' repair posture is for
246
+ // the operator CRUD path, where "missing" means "first trigger".
247
+ return { invalid: `triggers file unreadable (${err?.code ?? "parse-error"}), disarm not written: ${triggersPath}` };
248
+ }
249
+ const entries = Array.isArray(raw?.triggers) ? raw.triggers : null;
250
+ const entry = entries?.[index];
251
+ if (!entry || typeof entry !== "object") {
252
+ return { invalid: `no trigger at index ${index} -- the file changed since this job matched; not disarming a stranger` };
253
+ }
254
+ if (entry.on?.once !== true) {
255
+ return { invalid: `trigger at index ${index} is not an armed one-shot -- the file changed since this job matched; not disarming a stranger` };
256
+ }
257
+ if (entry.on?.number !== number) {
258
+ return { invalid: `trigger at index ${index} names item #${entry.on?.number}, this job matched #${number} -- a re-attributed index refuses rather than disarming a stranger` };
259
+ }
260
+ // Compared only when the caller supplied one, and BOTH lanes checked when it did: a job
261
+ // carries exactly one of flow/command, and the entry must agree on which lane as well as
262
+ // the name -- a flow job must not spend a command one-shot that took the same slot.
263
+ if (flow !== undefined && (entry.run?.flow !== flow || entry.run?.command !== undefined)) {
264
+ return { invalid: `trigger at index ${index} does not dispatch flow ${JSON.stringify(flow)} -- the entry changed since this job matched; not disarming a stranger` };
265
+ }
266
+ if (command !== undefined && (entry.run?.command !== command || entry.run?.flow !== undefined)) {
267
+ return { invalid: `trigger at index ${index} does not dispatch command ${JSON.stringify(command)} -- the entry changed since this job matched; not disarming a stranger` };
268
+ }
269
+ if (entry.on.disarmed !== undefined) {
270
+ return { already: true };
271
+ }
272
+ entry.on.disarmed = { at, ...(jobId !== undefined && { jobId }) };
273
+ const text = serialize(entries);
274
+ try {
275
+ parseTriggers(text, triggersPath); // the shared validator, same rule as every write
276
+ } catch (e) {
277
+ return { invalid: e?.message ?? String(e) };
278
+ }
279
+ const tmp = tmpPathFor(triggersPath);
280
+ try {
281
+ fs.writeFileSync(tmp, text, { mode: 0o644 });
282
+ renameIntoPlace(fs, tmp, triggersPath);
283
+ } catch (err) {
284
+ discardTmp(fs, tmp);
285
+ return { invalid: `triggers file write failed (${err?.code ?? "write-error"}): ${triggersPath}` };
286
+ }
287
+ return { ok: true };
288
+ } finally {
289
+ releaseLock(fd, triggersPath, fs, log);
290
+ }
291
+ }
292
+ return { invalid: `triggers file locked after ${DISARM_LOCK_ATTEMPTS} attempts: ${lockPathFor(triggersPath)}` };
293
+ }
294
+
295
+ /**
296
+ * The pre-spend read (worker slice of #231): what does the FILE currently say about the one-shot at
297
+ * `index`? Fail-open by design -- the caller refuses a job only on POSITIVE disarmed evidence, so
298
+ * "unknown" (unreadable file, index gone, entry no longer a one-shot) means "run": a broken read must
299
+ * never wedge every once job, and the identity mismatch cases are the disarm writer's to refuse.
300
+ */
301
+ /**
302
+ * The post-record disarm hook (issue #231): wired around the worker's one recordRun funnel, called
303
+ * strictly AFTER writeRecord returns, for EVERY record -- completed, policy, and per-attempt failed
304
+ * alike, because "fired" means "produced a run record" (the issue's own definition) and the
305
+ * pre-spend check's own-jobId exception is what keeps BullMQ's second attempt of the same delivery
306
+ * runnable. NEVER throws and never rejects: a disarm failure is a loud log line, not a crashed
307
+ * record path.
308
+ *
309
+ * `triggersPath` may be null only when even the cwd fallback could not be formed; the caller
310
+ * resolves `PI_TRIGGERS_FILE ?? join(cwd, "triggers.json")` -- doctor's own precedent, NOT the
311
+ * worker config's `triggersFile` (whose null means "cron disabled" and must keep meaning that;
312
+ * under that knob the DEFAULT single-host deployment would have a firing receiver and a worker
313
+ * that can neither disarm nor pre-spend-check).
314
+ */
315
+ export function makeDisarmOnce({ triggersPath, fs = nodeFs, log = () => {}, disarm = disarmTrigger }) {
316
+ return async function disarmOnce({ job, endedAt }) {
317
+ try {
318
+ const matched = job?.data?.trigger?.matched;
319
+ if (matched?.once !== true) return; // every unflagged job takes zero new code paths
320
+ const triggerIndex = matched.index ?? null;
321
+ if (typeof triggersPath !== "string" || triggersPath === "") {
322
+ log("trigger_disarm_unavailable", { jobId: job?.id ?? null, triggerIndex, reason: "triggers file unresolvable" });
323
+ return;
324
+ }
325
+ // The identity the writer re-checks: the item number (the issue shape carries it on matched,
326
+ // the PR shape on the target) and the dispatch lane the job actually carried.
327
+ const number = matched.number ?? job?.data?.target?.number;
328
+ const res = await disarm({
329
+ triggersPath,
330
+ index: matched.index,
331
+ number,
332
+ flow: job?.data?.flow,
333
+ command: job?.data?.command,
334
+ jobId: job?.id,
335
+ at: endedAt,
336
+ fs,
337
+ log,
338
+ });
339
+ if (res.ok) log("trigger_disarmed", { jobId: job?.id ?? null, triggerIndex });
340
+ else if (res.already) log("trigger_already_disarmed", { jobId: job?.id ?? null, triggerIndex });
341
+ else log("trigger_disarm_failed", { jobId: job?.id ?? null, triggerIndex, reason: res.invalid });
342
+ } catch (err) {
343
+ // Unreachable by construction (disarmTrigger never throws), kept because this hook sits on
344
+ // the record path and a record must never be lost to bookkeeping.
345
+ log("trigger_disarm_failed", { jobId: job?.id ?? null, triggerIndex: job?.data?.trigger?.matched?.index ?? null, reason: err?.code ?? "disarm-error" });
346
+ }
347
+ };
348
+ }
349
+
350
+ /**
351
+ * The pre-spend check's factory (issue #231): `(job, { queueJobId }) => { ok } | { refused, at, jobId }`.
352
+ * Refuses ONLY on positive FOREIGN disarmed evidence -- a mark whose jobId is this very queue job
353
+ * means BullMQ's second attempt of the delivery that spent the trigger, which must still run
354
+ * (without the exception, a disarm on attempt one's failure record silently turns attempts:2 into
355
+ * attempts:1 for every once job). A hand-written mark carries no jobId and reads as foreign, which
356
+ * is exactly what an operator disarming by hand intends. Everything else -- unreadable file, index
357
+ * gone, entry changed -- is "run": fail-open, the disarm writer owns the loud refusals, and in the
358
+ * compose topology (single-file :ro bind mount pinned to a dead inode, so the receiver never sees
359
+ * the disarm until restart) this check IS the once-enforcement layer, which is why it exists at all.
360
+ */
361
+ export function makeCheckOnceSpent({ triggersPath, fs = nodeFs }) {
362
+ return async function checkOnceSpent(job, { queueJobId } = {}) {
363
+ if (typeof triggersPath !== "string" || triggersPath === "") return { ok: true };
364
+ const matched = job?.trigger?.matched;
365
+ const state = readDisarmState({
366
+ triggersPath,
367
+ index: matched?.index,
368
+ number: matched?.number ?? job?.target?.number,
369
+ flow: job?.flow,
370
+ command: job?.command,
371
+ fs,
372
+ });
373
+ if (state.state !== "disarmed") return { ok: true };
374
+ if (state.jobId !== null && state.jobId === queueJobId) return { ok: true }; // our own earlier attempt
375
+ return { refused: true, at: state.at, jobId: state.jobId };
376
+ };
377
+ }
378
+
379
+ export function readDisarmState({ triggersPath, index, number, flow, command, fs = nodeFs }) {
380
+ let raw;
381
+ try {
382
+ raw = JSON.parse(fs.readFileSync(triggersPath, "utf8"));
383
+ } catch (err) {
384
+ return { state: "unknown", reason: err?.code ?? "parse-error" };
385
+ }
386
+ const entry = Array.isArray(raw?.triggers) ? raw.triggers[index] : undefined;
387
+ if (!entry || typeof entry !== "object" || entry.on?.once !== true || entry.on?.number !== number) {
388
+ return { state: "unknown", reason: "entry-changed" };
389
+ }
390
+ // The disarm writer's identity fields, folded to "unknown" rather than refused: a re-armed
391
+ // DIFFERENT one-shot at this index is not spent, so the fail-open answer -- run -- is the true one.
392
+ if (flow !== undefined && (entry.run?.flow !== flow || entry.run?.command !== undefined)) {
393
+ return { state: "unknown", reason: "entry-changed" };
394
+ }
395
+ if (command !== undefined && (entry.run?.command !== command || entry.run?.flow !== undefined)) {
396
+ return { state: "unknown", reason: "entry-changed" };
397
+ }
398
+ if (entry.on.disarmed !== undefined) {
399
+ const d = entry.on.disarmed;
400
+ return { state: "disarmed", at: typeof d?.at === "string" ? d.at : null, jobId: typeof d?.jobId === "string" ? d.jobId : null };
401
+ }
402
+ return { state: "armed" };
403
+ }