@edgehero/pi-dispatch 1.10.2 → 1.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.10.2",
3
+ "version": "1.10.3",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
@@ -194,9 +194,9 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
194
194
 
195
195
  /**
196
196
  * Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
197
- * `setTimeout` -- and `.unref()`'d so it can never hold the process open, which is the posture the
198
- * three `fs.watch` watchers already take. `stop` is registered as an extraCloser beside the runtime
199
- * queue, so a clean shutdown clears it before `process.exit`.
197
+ * `setTimeout` -- and `.unref()`'d so it can never hold the process open. `close` is registered as an
198
+ * extraCloser beside the runtime queue, so a clean shutdown clears it before `process.exit`; the three
199
+ * `fs.watch` watchers take the same two-part posture since issue #295, unref'd AND closed.
200
200
  */
201
201
  async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
202
202
  if (closed || timer) return; // a second start would leak the first interval
package/src/index.mjs CHANGED
@@ -836,10 +836,30 @@ export function createWorker({ connection, name, stopContainer, containerName, h
836
836
  // outlive the handler that was meant to stop them.
837
837
  for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
838
838
  for (const w of workers) await w.close().catch(() => {});
839
- // Close auxiliary resources (e.g. a cron scheduler) after the worker drains. Per-item catch
840
- // so one failing or absent closer never strands the others or blocks exit -- matches the
841
- // swallow posture on cancelAllJobs above.
842
- await Promise.all(extraClosers.map((c) => Promise.resolve(c.close?.()).catch(() => {})));
839
+ // Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
840
+ // Per-item catch so one failing or absent closer never strands the others or blocks exit -- matches
841
+ // the swallow posture on cancelAllJobs above. The try/catch is NOT redundant with the `.catch`:
842
+ // `Promise.resolve(x)` does not catch a SYNCHRONOUS throw from `x`, and `c.close` on a null entry
843
+ // throws before `Promise.resolve` is ever reached. Either would escape this callback, reject the whole
844
+ // shutdown and skip the `process.exit(0)` below. Jobs and containers are already stopped by then, so
845
+ // what a stranded loop leaks is the rest of the list: `registry.close()` is the DEL that keeps a
846
+ // stopped host from lingering as a ghost peer for its full TTL, and a ghost peer with a stale
847
+ // `fpCron` is what makes a later `reconcileGated` refuse a legitimate reconcile. The comment above
848
+ // promised this isolation before the code delivered it (issue #295). It bounds nothing, though: a
849
+ // closer that never settles still blocks exit, which no closer here does.
850
+ //
851
+ // Read LATE and deliberately: `start.mjs` pushes its live-edit watchers into this array AFTER handing
852
+ // it over, because they are armed after the boot reconcile. Anything here that snapshots or copies
853
+ // the array un-registers them in silence.
854
+ await Promise.all(
855
+ extraClosers.map((c) => {
856
+ try {
857
+ return Promise.resolve(c?.close?.()).catch(() => {});
858
+ } catch {
859
+ return Promise.resolve();
860
+ }
861
+ }),
862
+ );
843
863
  process.exit(0);
844
864
  };
845
865
  process.once("SIGTERM", shutdown);
package/src/start.mjs CHANGED
@@ -84,57 +84,144 @@ const WORKER_VERSION = (() => {
84
84
  * `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
85
85
  * `reaper_skipped`, and boot continues to the worker.
86
86
  */
87
+ /**
88
+ * The stop handle every live-edit watch below hands back, so `startWorker` can register it in the same
89
+ * `extraClosers` list that already closes the queues and the host registry (`index.mjs` -> shutdown).
90
+ *
91
+ * A WATCH NOTHING CAN CLOSE IS NOT A DETAIL (issue #295). `watch(dir, cb).unref?.()` retained nothing, so
92
+ * the watch outlived the worker that armed it, and the reload it later fired ran through THAT boot's
93
+ * `log` closure: that boot's injected `write`, stamped with that boot's `workerName`. One process running
94
+ * one worker, that is a rounding error at exit. One process running forty boots, which is what a test file
95
+ * is, and a worker that shut down two tests ago writes into a live worker's capture under a host that is
96
+ * not running -- `every log line carries the host` went red in CI reading `runnervmejwal` where it
97
+ * asserted `mac-mini-1`.
98
+ *
99
+ * UNREF'D IS NOT CLEANED UP, and that difference is what hid this across three features. `unref` says only
100
+ * that a handle will not hold the event loop open; the watch stays armed either way.
101
+ * `INT-HOST-REGISTRY-CONTRACT` states the same distinction from the opposite side, where a bound's timer
102
+ * is deliberately NOT unref'd because an unref'd timer does not fire when the hung command is the last
103
+ * thing holding the loop.
104
+ *
105
+ * Three properties, each one a way the shutdown breaks without it:
106
+ *
107
+ * - It CLOSES the FSWatcher, which is the leak itself.
108
+ * - It CANCELS the debounce the watcher already armed. Closing a watcher does not cancel a `setTimeout`
109
+ * the callback already set, and only the watcher was ever unref'd -- the 150ms timer never was. In a
110
+ * real worker that costs nothing, because the shutdown ends in `process.exit(0)` either way; it is the
111
+ * harness, where the loop is left to drain on its own, that the stray timer reaches.
112
+ * - Its `close()` NEVER THROWS for the handles its three callers build, and is idempotent -- by NULLING
113
+ * what it closed rather than by an early return, which would be a guard with nothing behind it. Node's
114
+ * own `FSWatcher.close()` is already both (measured: a second close returns early and neither throws),
115
+ * but this closer must not
116
+ * INHERIT that guarantee, it must MAKE it: the shutdown loop in `index.mjs` cannot catch a SYNCHRONOUS
117
+ * throw from a closer, and the comment there carries the argument. The swallow is
118
+ * `makeHostRegistry.close`'s posture rather than a new one.
119
+ *
120
+ * A watch that was never created -- the `catch` arm of each function below, a platform without `fs.watch`
121
+ * -- still gets a closer, so registration is unconditional and the list's shape never depends on the
122
+ * platform. That is why all three return from OUTSIDE their try/catch.
123
+ *
124
+ * WHAT IT CANNOT DO, because the list above would otherwise read as complete: cancel a reload that has
125
+ * ALREADY started. `reloadSchedules` is async and awaits a Valkey round trip, so a debounce that fired
126
+ * just before the close is still running after it -- and the watchers stay armed for the whole drain
127
+ * ahead of the closer loop, not merely 150ms. That reload cannot be recalled, so what is gated instead is
128
+ * its VOICE: `reloadLog` below goes quiet once `closed` is set, and every reload is handed that instead
129
+ * of the boot's own `log`, which is what the
130
+ * issue actually asks for -- a stopped worker writes no line. The reload's own Valkey work may still be
131
+ * cut off mid-flight by the queue closing beside it, leaving a scheduler set the next boot's reconcile
132
+ * repairs; that race predates this change and is not narrowed by it.
133
+ *
134
+ * EXPORTED for the reason `reloadScopedLimits` is: none of the three properties is observable through a
135
+ * real `fs.watch` without racing the filesystem, and a guarantee the shutdown rests on deserves a
136
+ * deterministic pin rather than a sleep.
137
+ */
138
+ export function makeWatchCloser(handles, log) {
139
+ return {
140
+ // The reload's voice, and the reason this factory is handed the boot's `log` rather than only its
141
+ // handles. A reload already in flight cannot be recalled, so what the close gates is what it can
142
+ // still SAY: after `closed`, a line from this watch would carry the host of a worker that has
143
+ // stopped, which is the bleed the issue is about. The arming lines keep the real `log` -- they run
144
+ // before any close.
145
+ reloadLog: (event, fields) => {
146
+ if (!handles.closed) log(event, fields);
147
+ },
148
+ close() {
149
+ // `closed` FIRST, before anything is torn down: it is what gates `reloadLog` above and the watch
150
+ // callback below, so a callback or a reload landing mid-close is already silenced.
151
+ handles.closed = true;
152
+ clearTimeout(handles.timer);
153
+ handles.timer = null;
154
+ try {
155
+ handles.watcher?.close();
156
+ } catch {
157
+ // A close that failed has already stopped mattering, and a THROW here rejects the shutdown.
158
+ }
159
+ handles.watcher = null;
160
+ },
161
+ };
162
+ }
163
+
87
164
  /**
88
165
  * Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
89
166
  * inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
90
- * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
91
- * logs and the worker keeps its boot-time schedulers.
167
+ * `reloadSchedules`. Best-effort: a platform without `fs.watch` logs and the worker keeps its boot-time
168
+ * schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
169
+ * `startWorker` registers so the watch dies with the worker that armed it (issue #295).
92
170
  */
93
171
  function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
94
172
  const path = config.triggersFile;
95
173
  const dir = dirname(path) || ".";
96
174
  const file = basename(path);
97
- let timer = null;
175
+ const handles = { watcher: null, timer: null, closed: false };
176
+ const closer = makeWatchCloser(handles, log);
98
177
  try {
99
- watch(dir, (_event, changed) => {
178
+ handles.watcher = watch(dir, (_event, changed) => {
179
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
100
180
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
101
- clearTimeout(timer);
102
- timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
103
- }).unref?.();
181
+ clearTimeout(handles.timer);
182
+ handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), 150);
183
+ });
184
+ handles.watcher.unref?.();
104
185
  log("triggers_watching", { path });
105
186
  } catch (err) {
106
187
  log("triggers_watch_unavailable", { reason: err?.message });
107
188
  }
189
+ return closer;
108
190
  }
109
191
 
110
192
  /**
111
193
  * Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
112
194
  * and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
113
- * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort + unref'd.
195
+ * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
196
+ * FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
114
197
  */
115
198
  function watchPauseWindowsFile(config, ref, log) {
116
199
  const path = config.pauseWindowsFile;
117
200
  const dir = dirname(path) || ".";
118
201
  const file = basename(path);
119
- let timer = null;
202
+ const handles = { watcher: null, timer: null, closed: false };
203
+ const closer = makeWatchCloser(handles, log);
120
204
  const reload = () => {
121
205
  try {
122
206
  ref.current = loadPauseWindows(config);
123
- log("pause_windows_reloaded", { count: ref.current.length });
207
+ closer.reloadLog("pause_windows_reloaded", { count: ref.current.length });
124
208
  } catch (err) {
125
- log("pause_windows_reload_invalid", { reason: err?.message });
209
+ closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
126
210
  }
127
211
  };
128
212
  try {
129
- watch(dir, (_event, changed) => {
213
+ handles.watcher = watch(dir, (_event, changed) => {
214
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
130
215
  if (changed && changed !== file) return;
131
- clearTimeout(timer);
132
- timer = setTimeout(reload, 150);
133
- }).unref?.();
216
+ clearTimeout(handles.timer);
217
+ handles.timer = setTimeout(reload, 150);
218
+ });
219
+ handles.watcher.unref?.();
134
220
  log("pause_windows_watching", { path });
135
221
  } catch (err) {
136
222
  log("pause_windows_watch_unavailable", { reason: err?.message });
137
223
  }
224
+ return closer;
138
225
  }
139
226
 
140
227
  /**
@@ -154,23 +241,28 @@ export function reloadScopedLimits(config, ref, log) {
154
241
 
155
242
  /**
156
243
  * Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
157
- * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort + unref'd.
244
+ * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
245
+ * unref'd and the returned closer stops the watch with the worker (issue #295).
158
246
  */
159
247
  function watchScopedLimitsFile(config, ref, log) {
160
248
  const path = config.scopedLimitsFile;
161
249
  const dir = dirname(path) || ".";
162
250
  const file = basename(path);
163
- let timer = null;
251
+ const handles = { watcher: null, timer: null, closed: false };
252
+ const closer = makeWatchCloser(handles, log);
164
253
  try {
165
- watch(dir, (_event, changed) => {
254
+ handles.watcher = watch(dir, (_event, changed) => {
255
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
166
256
  if (changed && changed !== file) return;
167
- clearTimeout(timer);
168
- timer = setTimeout(() => reloadScopedLimits(config, ref, log), 150);
169
- }).unref?.();
257
+ clearTimeout(handles.timer);
258
+ handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), 150);
259
+ });
260
+ handles.watcher.unref?.();
170
261
  log("scoped_limits_watching", { path });
171
262
  } catch (err) {
172
263
  log("scoped_limits_watch_unavailable", { reason: err?.message });
173
264
  }
265
+ return closer;
174
266
  }
175
267
 
176
268
  /**
@@ -677,6 +769,18 @@ export async function startWorker(
677
769
  reaps: backendReaps,
678
770
  });
679
771
 
772
+ // The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
773
+ // array rather than the literal it used to be, because up to three of its members do not exist yet: the
774
+ // live-edit watchers are armed at the END of boot, below, and deliberately after the boot reconcile --
775
+ // arming them earlier would let an operator edit run `reloadSchedules` concurrently with the boot
776
+ // `reconcileGated`, on a different queue handle, and reconcile's orphan prune is not safe against that.
777
+ //
778
+ // PUSHING AFTER THE HANDOFF IS SOUND FOR ONE REASON ONLY: `index.mjs` reads this array at SHUTDOWN time,
779
+ // not when it receives it, and so does the test harness at teardown. A refactor that COPIES it there --
780
+ // a spread, a freeze, a snapshot inside `createWorker` -- un-registers the watchers in SILENCE and puts
781
+ // issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
782
+ const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
783
+
680
784
  const worker = createWorkerFn({
681
785
  connection: parseConnection(config.valkeyUrl),
682
786
  // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
@@ -696,7 +800,7 @@ export async function startWorker(
696
800
  getSettings,
697
801
  redis,
698
802
  recordRun,
699
- extraClosers: [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])],
803
+ extraClosers,
700
804
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
701
805
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
702
806
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
@@ -931,20 +1035,21 @@ export async function startWorker(
931
1035
 
932
1036
  // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
933
1037
  // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
934
- // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
1038
+ // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1039
+ // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
935
1040
  if (config.triggersFile) {
936
- watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
1041
+ extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared));
937
1042
  }
938
1043
 
939
1044
  // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
940
1045
  // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
941
1046
  if (config.pauseWindowsFile) {
942
- watchPauseWindowsFile(config, pauseWindows, log);
1047
+ extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log));
943
1048
  }
944
1049
 
945
1050
  // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
946
1051
  if (config.scopedLimitsFile) {
947
- watchScopedLimitsFile(config, scopedLimits, log);
1052
+ extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log));
948
1053
  }
949
1054
 
950
1055
  log("worker_started", {