@edgehero/pi-dispatch 1.10.2 → 1.10.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/host-registry.mjs +3 -3
- package/src/index.mjs +24 -4
- package/src/start.mjs +131 -26
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.10.
|
|
3
|
+
"version": "1.10.3",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
package/src/host-registry.mjs
CHANGED
|
@@ -194,9 +194,9 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
|
|
|
194
194
|
|
|
195
195
|
/**
|
|
196
196
|
* Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
|
|
197
|
-
* `setTimeout` -- and `.unref()`'d so it can never hold the process open
|
|
198
|
-
*
|
|
199
|
-
*
|
|
197
|
+
* `setTimeout` -- and `.unref()`'d so it can never hold the process open. `close` is registered as an
|
|
198
|
+
* extraCloser beside the runtime queue, so a clean shutdown clears it before `process.exit`; the three
|
|
199
|
+
* `fs.watch` watchers take the same two-part posture since issue #295, unref'd AND closed.
|
|
200
200
|
*/
|
|
201
201
|
async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
|
|
202
202
|
if (closed || timer) return; // a second start would leak the first interval
|
package/src/index.mjs
CHANGED
|
@@ -836,10 +836,30 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
836
836
|
// outlive the handler that was meant to stop them.
|
|
837
837
|
for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
|
|
838
838
|
for (const w of workers) await w.close().catch(() => {});
|
|
839
|
-
// Close auxiliary resources (
|
|
840
|
-
// so one failing or absent closer never strands the others or blocks exit -- matches
|
|
841
|
-
// swallow posture on cancelAllJobs above.
|
|
842
|
-
|
|
839
|
+
// Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
|
|
840
|
+
// Per-item catch so one failing or absent closer never strands the others or blocks exit -- matches
|
|
841
|
+
// the swallow posture on cancelAllJobs above. The try/catch is NOT redundant with the `.catch`:
|
|
842
|
+
// `Promise.resolve(x)` does not catch a SYNCHRONOUS throw from `x`, and `c.close` on a null entry
|
|
843
|
+
// throws before `Promise.resolve` is ever reached. Either would escape this callback, reject the whole
|
|
844
|
+
// shutdown and skip the `process.exit(0)` below. Jobs and containers are already stopped by then, so
|
|
845
|
+
// what a stranded loop leaks is the rest of the list: `registry.close()` is the DEL that keeps a
|
|
846
|
+
// stopped host from lingering as a ghost peer for its full TTL, and a ghost peer with a stale
|
|
847
|
+
// `fpCron` is what makes a later `reconcileGated` refuse a legitimate reconcile. The comment above
|
|
848
|
+
// promised this isolation before the code delivered it (issue #295). It bounds nothing, though: a
|
|
849
|
+
// closer that never settles still blocks exit, which no closer here does.
|
|
850
|
+
//
|
|
851
|
+
// Read LATE and deliberately: `start.mjs` pushes its live-edit watchers into this array AFTER handing
|
|
852
|
+
// it over, because they are armed after the boot reconcile. Anything here that snapshots or copies
|
|
853
|
+
// the array un-registers them in silence.
|
|
854
|
+
await Promise.all(
|
|
855
|
+
extraClosers.map((c) => {
|
|
856
|
+
try {
|
|
857
|
+
return Promise.resolve(c?.close?.()).catch(() => {});
|
|
858
|
+
} catch {
|
|
859
|
+
return Promise.resolve();
|
|
860
|
+
}
|
|
861
|
+
}),
|
|
862
|
+
);
|
|
843
863
|
process.exit(0);
|
|
844
864
|
};
|
|
845
865
|
process.once("SIGTERM", shutdown);
|
package/src/start.mjs
CHANGED
|
@@ -84,57 +84,144 @@ const WORKER_VERSION = (() => {
|
|
|
84
84
|
* `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
|
|
85
85
|
* `reaper_skipped`, and boot continues to the worker.
|
|
86
86
|
*/
|
|
87
|
+
/**
|
|
88
|
+
* The stop handle every live-edit watch below hands back, so `startWorker` can register it in the same
|
|
89
|
+
* `extraClosers` list that already closes the queues and the host registry (`index.mjs` -> shutdown).
|
|
90
|
+
*
|
|
91
|
+
* A WATCH NOTHING CAN CLOSE IS NOT A DETAIL (issue #295). `watch(dir, cb).unref?.()` retained nothing, so
|
|
92
|
+
* the watch outlived the worker that armed it, and the reload it later fired ran through THAT boot's
|
|
93
|
+
* `log` closure: that boot's injected `write`, stamped with that boot's `workerName`. One process running
|
|
94
|
+
* one worker, that is a rounding error at exit. One process running forty boots, which is what a test file
|
|
95
|
+
* is, and a worker that shut down two tests ago writes into a live worker's capture under a host that is
|
|
96
|
+
* not running -- `every log line carries the host` went red in CI reading `runnervmejwal` where it
|
|
97
|
+
* asserted `mac-mini-1`.
|
|
98
|
+
*
|
|
99
|
+
* UNREF'D IS NOT CLEANED UP, and that difference is what hid this across three features. `unref` says only
|
|
100
|
+
* that a handle will not hold the event loop open; the watch stays armed either way.
|
|
101
|
+
* `INT-HOST-REGISTRY-CONTRACT` states the same distinction from the opposite side, where a bound's timer
|
|
102
|
+
* is deliberately NOT unref'd because an unref'd timer does not fire when the hung command is the last
|
|
103
|
+
* thing holding the loop.
|
|
104
|
+
*
|
|
105
|
+
* Three properties, each one a way the shutdown breaks without it:
|
|
106
|
+
*
|
|
107
|
+
* - It CLOSES the FSWatcher, which is the leak itself.
|
|
108
|
+
* - It CANCELS the debounce the watcher already armed. Closing a watcher does not cancel a `setTimeout`
|
|
109
|
+
* the callback already set, and only the watcher was ever unref'd -- the 150ms timer never was. In a
|
|
110
|
+
* real worker that costs nothing, because the shutdown ends in `process.exit(0)` either way; it is the
|
|
111
|
+
* harness, where the loop is left to drain on its own, that the stray timer reaches.
|
|
112
|
+
* - Its `close()` NEVER THROWS for the handles its three callers build, and is idempotent -- by NULLING
|
|
113
|
+
* what it closed rather than by an early return, which would be a guard with nothing behind it. Node's
|
|
114
|
+
* own `FSWatcher.close()` is already both (measured: a second close returns early and neither throws),
|
|
115
|
+
* but this closer must not
|
|
116
|
+
* INHERIT that guarantee, it must MAKE it: the shutdown loop in `index.mjs` cannot catch a SYNCHRONOUS
|
|
117
|
+
* throw from a closer, and the comment there carries the argument. The swallow is
|
|
118
|
+
* `makeHostRegistry.close`'s posture rather than a new one.
|
|
119
|
+
*
|
|
120
|
+
* A watch that was never created -- the `catch` arm of each function below, a platform without `fs.watch`
|
|
121
|
+
* -- still gets a closer, so registration is unconditional and the list's shape never depends on the
|
|
122
|
+
* platform. That is why all three return from OUTSIDE their try/catch.
|
|
123
|
+
*
|
|
124
|
+
* WHAT IT CANNOT DO, because the list above would otherwise read as complete: cancel a reload that has
|
|
125
|
+
* ALREADY started. `reloadSchedules` is async and awaits a Valkey round trip, so a debounce that fired
|
|
126
|
+
* just before the close is still running after it -- and the watchers stay armed for the whole drain
|
|
127
|
+
* ahead of the closer loop, not merely 150ms. That reload cannot be recalled, so what is gated instead is
|
|
128
|
+
* its VOICE: `reloadLog` below goes quiet once `closed` is set, and every reload is handed that instead
|
|
129
|
+
* of the boot's own `log`, which is what the
|
|
130
|
+
* issue actually asks for -- a stopped worker writes no line. The reload's own Valkey work may still be
|
|
131
|
+
* cut off mid-flight by the queue closing beside it, leaving a scheduler set the next boot's reconcile
|
|
132
|
+
* repairs; that race predates this change and is not narrowed by it.
|
|
133
|
+
*
|
|
134
|
+
* EXPORTED for the reason `reloadScopedLimits` is: none of the three properties is observable through a
|
|
135
|
+
* real `fs.watch` without racing the filesystem, and a guarantee the shutdown rests on deserves a
|
|
136
|
+
* deterministic pin rather than a sleep.
|
|
137
|
+
*/
|
|
138
|
+
export function makeWatchCloser(handles, log) {
|
|
139
|
+
return {
|
|
140
|
+
// The reload's voice, and the reason this factory is handed the boot's `log` rather than only its
|
|
141
|
+
// handles. A reload already in flight cannot be recalled, so what the close gates is what it can
|
|
142
|
+
// still SAY: after `closed`, a line from this watch would carry the host of a worker that has
|
|
143
|
+
// stopped, which is the bleed the issue is about. The arming lines keep the real `log` -- they run
|
|
144
|
+
// before any close.
|
|
145
|
+
reloadLog: (event, fields) => {
|
|
146
|
+
if (!handles.closed) log(event, fields);
|
|
147
|
+
},
|
|
148
|
+
close() {
|
|
149
|
+
// `closed` FIRST, before anything is torn down: it is what gates `reloadLog` above and the watch
|
|
150
|
+
// callback below, so a callback or a reload landing mid-close is already silenced.
|
|
151
|
+
handles.closed = true;
|
|
152
|
+
clearTimeout(handles.timer);
|
|
153
|
+
handles.timer = null;
|
|
154
|
+
try {
|
|
155
|
+
handles.watcher?.close();
|
|
156
|
+
} catch {
|
|
157
|
+
// A close that failed has already stopped mattering, and a THROW here rejects the shutdown.
|
|
158
|
+
}
|
|
159
|
+
handles.watcher = null;
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
87
164
|
/**
|
|
88
165
|
* Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
|
|
89
166
|
* inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
|
|
90
|
-
* `reloadSchedules`. Best-effort
|
|
91
|
-
*
|
|
167
|
+
* `reloadSchedules`. Best-effort: a platform without `fs.watch` logs and the worker keeps its boot-time
|
|
168
|
+
* schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
|
|
169
|
+
* `startWorker` registers so the watch dies with the worker that armed it (issue #295).
|
|
92
170
|
*/
|
|
93
171
|
function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
|
|
94
172
|
const path = config.triggersFile;
|
|
95
173
|
const dir = dirname(path) || ".";
|
|
96
174
|
const file = basename(path);
|
|
97
|
-
|
|
175
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
176
|
+
const closer = makeWatchCloser(handles, log);
|
|
98
177
|
try {
|
|
99
|
-
watch(dir, (_event, changed) => {
|
|
178
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
179
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
100
180
|
if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
|
|
101
|
-
clearTimeout(timer);
|
|
102
|
-
timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
|
|
103
|
-
})
|
|
181
|
+
clearTimeout(handles.timer);
|
|
182
|
+
handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), 150);
|
|
183
|
+
});
|
|
184
|
+
handles.watcher.unref?.();
|
|
104
185
|
log("triggers_watching", { path });
|
|
105
186
|
} catch (err) {
|
|
106
187
|
log("triggers_watch_unavailable", { reason: err?.message });
|
|
107
188
|
}
|
|
189
|
+
return closer;
|
|
108
190
|
}
|
|
109
191
|
|
|
110
192
|
/**
|
|
111
193
|
* Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
|
|
112
194
|
* and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
|
|
113
|
-
* effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort
|
|
195
|
+
* effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
|
|
196
|
+
* FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
|
|
114
197
|
*/
|
|
115
198
|
function watchPauseWindowsFile(config, ref, log) {
|
|
116
199
|
const path = config.pauseWindowsFile;
|
|
117
200
|
const dir = dirname(path) || ".";
|
|
118
201
|
const file = basename(path);
|
|
119
|
-
|
|
202
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
203
|
+
const closer = makeWatchCloser(handles, log);
|
|
120
204
|
const reload = () => {
|
|
121
205
|
try {
|
|
122
206
|
ref.current = loadPauseWindows(config);
|
|
123
|
-
|
|
207
|
+
closer.reloadLog("pause_windows_reloaded", { count: ref.current.length });
|
|
124
208
|
} catch (err) {
|
|
125
|
-
|
|
209
|
+
closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
|
|
126
210
|
}
|
|
127
211
|
};
|
|
128
212
|
try {
|
|
129
|
-
watch(dir, (_event, changed) => {
|
|
213
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
214
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
130
215
|
if (changed && changed !== file) return;
|
|
131
|
-
clearTimeout(timer);
|
|
132
|
-
timer = setTimeout(reload, 150);
|
|
133
|
-
})
|
|
216
|
+
clearTimeout(handles.timer);
|
|
217
|
+
handles.timer = setTimeout(reload, 150);
|
|
218
|
+
});
|
|
219
|
+
handles.watcher.unref?.();
|
|
134
220
|
log("pause_windows_watching", { path });
|
|
135
221
|
} catch (err) {
|
|
136
222
|
log("pause_windows_watch_unavailable", { reason: err?.message });
|
|
137
223
|
}
|
|
224
|
+
return closer;
|
|
138
225
|
}
|
|
139
226
|
|
|
140
227
|
/**
|
|
@@ -154,23 +241,28 @@ export function reloadScopedLimits(config, ref, log) {
|
|
|
154
241
|
|
|
155
242
|
/**
|
|
156
243
|
* Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
|
|
157
|
-
* for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort
|
|
244
|
+
* for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
|
|
245
|
+
* unref'd and the returned closer stops the watch with the worker (issue #295).
|
|
158
246
|
*/
|
|
159
247
|
function watchScopedLimitsFile(config, ref, log) {
|
|
160
248
|
const path = config.scopedLimitsFile;
|
|
161
249
|
const dir = dirname(path) || ".";
|
|
162
250
|
const file = basename(path);
|
|
163
|
-
|
|
251
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
252
|
+
const closer = makeWatchCloser(handles, log);
|
|
164
253
|
try {
|
|
165
|
-
watch(dir, (_event, changed) => {
|
|
254
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
255
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
166
256
|
if (changed && changed !== file) return;
|
|
167
|
-
clearTimeout(timer);
|
|
168
|
-
timer = setTimeout(() => reloadScopedLimits(config, ref,
|
|
169
|
-
})
|
|
257
|
+
clearTimeout(handles.timer);
|
|
258
|
+
handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), 150);
|
|
259
|
+
});
|
|
260
|
+
handles.watcher.unref?.();
|
|
170
261
|
log("scoped_limits_watching", { path });
|
|
171
262
|
} catch (err) {
|
|
172
263
|
log("scoped_limits_watch_unavailable", { reason: err?.message });
|
|
173
264
|
}
|
|
265
|
+
return closer;
|
|
174
266
|
}
|
|
175
267
|
|
|
176
268
|
/**
|
|
@@ -677,6 +769,18 @@ export async function startWorker(
|
|
|
677
769
|
reaps: backendReaps,
|
|
678
770
|
});
|
|
679
771
|
|
|
772
|
+
// The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
|
|
773
|
+
// array rather than the literal it used to be, because up to three of its members do not exist yet: the
|
|
774
|
+
// live-edit watchers are armed at the END of boot, below, and deliberately after the boot reconcile --
|
|
775
|
+
// arming them earlier would let an operator edit run `reloadSchedules` concurrently with the boot
|
|
776
|
+
// `reconcileGated`, on a different queue handle, and reconcile's orphan prune is not safe against that.
|
|
777
|
+
//
|
|
778
|
+
// PUSHING AFTER THE HANDOFF IS SOUND FOR ONE REASON ONLY: `index.mjs` reads this array at SHUTDOWN time,
|
|
779
|
+
// not when it receives it, and so does the test harness at teardown. A refactor that COPIES it there --
|
|
780
|
+
// a spread, a freeze, a snapshot inside `createWorker` -- un-registers the watchers in SILENCE and puts
|
|
781
|
+
// issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
|
|
782
|
+
const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
|
|
783
|
+
|
|
680
784
|
const worker = createWorkerFn({
|
|
681
785
|
connection: parseConnection(config.valkeyUrl),
|
|
682
786
|
// #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
|
|
@@ -696,7 +800,7 @@ export async function startWorker(
|
|
|
696
800
|
getSettings,
|
|
697
801
|
redis,
|
|
698
802
|
recordRun,
|
|
699
|
-
extraClosers
|
|
803
|
+
extraClosers,
|
|
700
804
|
// REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
|
|
701
805
|
// Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
|
|
702
806
|
pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
|
|
@@ -931,20 +1035,21 @@ export async function startWorker(
|
|
|
931
1035
|
|
|
932
1036
|
// DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
|
|
933
1037
|
// on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
|
|
934
|
-
// Only when a triggers file is configured; best-effort
|
|
1038
|
+
// Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
|
|
1039
|
+
// closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
|
|
935
1040
|
if (config.triggersFile) {
|
|
936
|
-
watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
|
|
1041
|
+
extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared));
|
|
937
1042
|
}
|
|
938
1043
|
|
|
939
1044
|
// REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
|
|
940
1045
|
// an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
|
|
941
1046
|
if (config.pauseWindowsFile) {
|
|
942
|
-
watchPauseWindowsFile(config, pauseWindows, log);
|
|
1047
|
+
extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log));
|
|
943
1048
|
}
|
|
944
1049
|
|
|
945
1050
|
// Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
|
|
946
1051
|
if (config.scopedLimitsFile) {
|
|
947
|
-
watchScopedLimitsFile(config, scopedLimits, log);
|
|
1052
|
+
extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log));
|
|
948
1053
|
}
|
|
949
1054
|
|
|
950
1055
|
log("worker_started", {
|