@edgehero/pi-dispatch 1.10.1 → 1.10.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/.env.example CHANGED
@@ -86,12 +86,22 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
86
86
  # PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
87
87
  # Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
88
88
  # PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
89
+ # PI_DISPATCH_DEPLOYMENT_FILE= # ABSOLUTE path to the deployment pointer the /dispatch panel reads to find a deployment built elsewhere
90
+ # (default: <your pi agent dir>/pi-dispatch-deployment.json). Read by the ADMIN EXTENSION only; the worker and receiver never look at it.
91
+ # Your own environment still wins key by key, so this points the panel at a deployment, it does not override one
92
+ # PI_GRAPH_DIR= # where `/dispatch insights` writes its HTML artifact (default: under the OS temp dir). Admin extension only. See docs/insights.md
89
93
 
90
94
  # --- Reuse your existing pi setup in every job (see docs/global-pi-overlay.md) ---
91
95
  # PI_GLOBAL_PI_DIR= # dir with your host pi setup (models.json/skills/APPEND_SYSTEM.md), mounted /opt/pi-global:ro into every job, layered UNDER each repo's .pi/. Unset = off. Stage it with: pi-dispatch import-pi
92
96
  # PI_GLOBAL_ALLOW_EXTENSIONS= # the overlay's extensions LOAD by default (staging them with import-pi, which prints each one, is the vetting step). Set exactly 0 to keep them staged but dormant.
93
97
  # Unset, empty and the legacy 1 all mean LOAD. ANY other value refuses to boot -- a typo must never silently leave code running against adversarial input with open egress.
94
98
  # This knob covers the OVERLAY only. A serviced repo's own /workspace/.pi/extensions load regardless (they are default-branch, merge-gated) -- see SECURITY.md.
99
+ # PI_CODING_AGENT_DIR= # where YOUR pi setup lives on this host (default: ~/.pi/agent). `pi-dispatch import-pi` reads its models.json,
100
+ # skills and APPEND_SYSTEM.md from here, and the panel looks here for the deployment pointer.
101
+ # It is also read AT JOB TIME: with no provider key in the environment the worker reads this directory's auth.json
102
+ # for one (on by default; PI_AUTH_FROM_PI=0 turns it off), so pointing this at the wrong place makes every job
103
+ # refuse pre-spend with no credential for the provider. It is the SOURCE that gets staged, never the thing mounted:
104
+ # PI_GLOBAL_PI_DIR above is what a job actually sees. Set it if your pi lives somewhere other than your home directory
95
105
  # PI_PACKAGES_FILE= # path to pi-packages.json (default: ./pi-packages.json; --packages-file wins). Read ONLY by `pi-dispatch import-pi --with-packages`, never at job time.
96
106
  # Staged packages/ rides INSIDE PI_GLOBAL_PI_DIR -- no separate mount, no separate env dir -- and loads for EVERY job once staged; decline it PER TRIGGER with "packages": false in triggers.json, NOT by an env flag.
97
107
  # Versions must be EXACT (no ^ ~ * or latest); staging uses --ignore-scripts, so a package needing a build step is staged INCOMPLETE and import-pi warns.
@@ -166,6 +176,12 @@ RECEIVER_BIND=0.0.0.0
166
176
  GITHUB_AUTH_SOURCE=gh
167
177
  # For GITHUB_AUTH_SOURCE=pat: a repo-scoped, short-expiry fine-grained PAT
168
178
  GITHUB_PAT=
179
+ # GITHUB_PAT_VAR= # which variable above actually holds the PAT. Default GITHUB_PAT; set this only if your
180
+ # secrets manager insists on its own name and you would rather not copy the value to a second key.
181
+ # The NAME is not checked against anything: whatever you put here is read verbatim, so a typo
182
+ # reads an empty variable and the worker refuses at boot naming the name you chose. Pointing it
183
+ # at a variable that holds something else (GITHUB_APP_PRIVATE_KEY, say) is the mistake worth
184
+ # knowing about, because nothing stops it and the PAT path would then send that value to GitHub.
169
185
  # For GITHUB_AUTH_SOURCE=app (optional; required for multi-tenant)
170
186
  # `pi-dispatch setup github` fills all three in one browser click (App Manifest flow) and writes the PEM 0600
171
187
  GITHUB_APP_ID=
@@ -176,6 +192,23 @@ GITHUB_APP_PRIVATE_KEY_PATH=
176
192
  # escapes both work. Never list it in PI_FORWARD_ENV: it mints tokens for every repo the App is on.
177
193
  GITHUB_APP_PRIVATE_KEY=
178
194
 
195
+ # --- Polling ingest, instead of a webhook (GitHub only) --- issue #282
196
+ # `pi-dispatch-receiver poll` fetches issue events, comments and pull requests over TLS with your own
197
+ # credential, so a deployment with no public URL, no DNS and no tunnel still fires triggers. Same gates and
198
+ # same queue as the webhook path; about a minute of latency instead of a second. Nothing here is read by
199
+ # `pi-dispatch-receiver serve`, and no WEBHOOK_SECRET is needed to poll: there is no inbound delivery to
200
+ # verify, because the poller originates every request itself. See docs/polling.md.
201
+ # POLL_REPOS= # WHICH repos to watch: comma-separated owner/name (e.g. acme/web,acme/api). Duplicates are dropped.
202
+ # Each entry must be exactly owner/name -- one slash, no spaces -- or the receiver refuses at boot naming the bad entry.
203
+ # Leave it UNSET only with GITHUB_AUTH_SOURCE=app: the poller then lists the App installation's own repos and
204
+ # re-lists every tenth cycle, so installing the App on a new repo starts polling it without an edit here.
205
+ # Unset under any other auth source is a boot refusal, deliberately: a PAT names no repo set, and a poller
206
+ # watching nothing looks exactly like a poller that is working.
207
+ # POLL_INTERVAL_SECONDS= # seconds between cycles. Default 60, floored at 30: a positive value BELOW the floor is raised to it,
208
+ # while 0, a negative, a fraction or junk still refuses at boot.
209
+ # GitHub asks pollers to respect its own x-poll-interval hint, which is honored as a MINIMUM when it arrives,
210
+ # so a busy hour slows the loop down rather than the loop hammering the API. A typo'd 1 must not turn this into a hammer.
211
+
179
212
  # --- GitLab trigger (receiver + worker auth) ---
180
213
  # Optional. Set these only to service GitLab projects; leaving GITLAB_TOKEN unset means no /gitlab
181
214
  # endpoint exists at all, rather than one that answers 401. See docs/gitlab.md.
@@ -184,6 +217,13 @@ GITHUB_APP_PRIVATE_KEY=
184
217
  # scope that can post a note -- GitLab offers no contents-vs-issues split -- so scope it to one project
185
218
  # and rotate it (CONST-TOKEN-SCOPED-PER-JOB). A GROUP token reaches every project in the group.
186
219
  GITLAB_TOKEN=
220
+ # GITLAB_AUTH_SOURCE= # accepts exactly one value, "pat", which is also the default, so there is nothing to set here.
221
+ # It exists to REFUSE the wrong assumption rather than to offer a choice: GITHUB_AUTH_SOURCE has
222
+ # three sources, and an operator who reasons by symmetry and writes app here gets a sentence at
223
+ # boot saying GitLab has no App equivalent, instead of a knob that is silently ignored.
224
+ # The refusal needs GITLAB_TOKEN to be set: with no token there is no GitLab to configure, the
225
+ # whole block is skipped, and a stray app here really is ignored. Same for the two below.
226
+ # FORGEJO_AUTH_SOURCE and AZURE_AUTH_SOURCE are the same variable for the same reason.
187
227
  # Your instance root. Only for self-hosted GitLab.
188
228
  GITLAB_URL=https://gitlab.com
189
229
  # How the receiver verifies a delivery. REQUIRED once any GITLAB_* variable is set, and deliberately not
@@ -205,6 +245,7 @@ FORGEJO_URL=
205
245
  # expire: there is no App or installation token, so rotation is the whole mitigation
206
246
  # (CONST-TOKEN-SCOPED-PER-JOB).
207
247
  FORGEJO_TOKEN=
248
+ # FORGEJO_AUTH_SOURCE= # only "pat", the default. See GITLAB_AUTH_SOURCE above for why it exists at all.
208
249
  # The harness account's NUMERIC id. Required when the token above is repository-scoped, because such a
209
250
  # token may not carry read:user and therefore cannot call GET /user. The receiver refuses to boot without an
210
251
  # identity from one source or the other: the bot-loop guard compares against it, and an unresolved identity
@@ -223,6 +264,7 @@ AZURE_ORG_URL=
223
264
  # permissions in Project Settings -- not from the token's scopes. It also needs vso.graph, to resolve the
224
265
  # actor's project membership before a job may be enqueued.
225
266
  AZURE_TOKEN=
267
+ # AZURE_AUTH_SOURCE= # only "pat", the default. See GITLAB_AUTH_SOURCE above for why it exists at all.
226
268
  # REQUIRED once any AZURE_* variable is set, and deliberately not defaulted: both modes are shared-secret
227
269
  # compares that cover no bytes, so which header carries the secret must be a choice somebody made.
228
270
  # basic -- Authorization: Basic <base64>, the credential you set on the subscription
@@ -232,3 +274,41 @@ AZURE_WEBHOOK_MODE=
232
274
  AZURE_WEBHOOK_SECRET=
233
275
  # Required only when AZURE_WEBHOOK_MODE=header.
234
276
  AZURE_WEBHOOK_HEADER=
277
+
278
+ # --- What is deliberately NOT a key in this file --- issue #282
279
+ # The rule, and it is checked by a test (worker/test/env-docs.test.mjs): every environment variable this
280
+ # project's loaders read is either a key above, or named here with the reason it cannot be one. A variable
281
+ # that is read and appears in neither is the failure this section exists to prevent, because an operator has
282
+ # no way to discover it and no way to find out that setting it did nothing.
283
+ #
284
+ # The worker's own inputs to a job container: PI_JOB_ID, PI_FLOW, PI_COMMAND, PI_PACKAGES, PI_SESSION_FILE,
285
+ # PI_OFFLINE, and the three PLAYWRIGHT_ names. The container's environment is BUILT, not inherited: the worker
286
+ # passes exactly the map it composed, so a value set here never reaches a job to be overridden in the first
287
+ # place. Several of them are also conditional, present only when the job has a flow, a command, staged
288
+ # packages or a session to resume. These are the container's side of INT-CONTAINER-RUNTIME-CONTRACT
289
+ # (specs/interfaces.md, docs/job-image.md), and run.secrets refuses the names at load so a trigger cannot
290
+ # bind one either. PI_FORWARD_ENV is NOT checked against these names, and it is applied after the worker's
291
+ # own map, so naming one there does override it. Not everything is: run.secrets, the egress variables and
292
+ # the minted forge token are all written later still and win over a forwarded value. A real edge, not a
293
+ # recommendation.
294
+ #
295
+ # PI_ENV_SETUP is an argument to `pi-dispatch service render|install --env-setup <absolute path>`, never a
296
+ # key. The service wrappers capture it BEFORE they source ./.env, precisely so that anything able to write
297
+ # this file cannot name a script the wrapper will run as the worker. A line here is honored by nothing, and
298
+ # that is the point rather than an oversight (docs/secrets.md, REQ-DEPLOYMENT-BOOTSTRAP).
299
+ #
300
+ # PI_RETRY_MAX (default 2) and PI_RETRY_BASE_MS (default 2000) are read by the runner INSIDE the container,
301
+ # and nothing on the host writes them, so a line here sets them on this machine and never reaches a job.
302
+ # PI_FORWARD_ENV is what carries a host value into a container, and is how you would actually change them.
303
+ #
304
+ # Provider key names are pi's, not ours. The worker asks pi which variable your PI_PROVIDER expects, so the
305
+ # provider block at the top of this file lists the common ones as examples and is not the whole set: pi
306
+ # supports around thirty providers and the answer travels with pi rather than with this file.
307
+ #
308
+ # Read from the surrounding system, not from a deployment: TMPDIR and TEMP (where the default job, log,
309
+ # graph and settings paths go), USER (the account a rendered service unit runs as), and TERM, SSH_CONNECTION,
310
+ # SSH_TTY, DISPLAY, WAYLAND_DISPLAY (how the panel decides whether it can open a browser for you).
311
+ #
312
+ # PI_DISPATCH_REQUIRE_LOADER_TESTS, PI_DISPATCH_REQUIRE_WORKER_TESTS, PI_DISPATCH_REQUIRE_RECEIVER_TESTS and
313
+ # VALKEY_TEST_URL turn locally skipped integration tests into required ones. They are read by the test files
314
+ # themselves and by CI, never by the worker, the receiver or the panel.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@edgehero/pi-dispatch",
3
- "version": "1.10.1",
3
+ "version": "1.10.3",
4
4
  "type": "module",
5
5
  "description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
6
6
  "keywords": [
package/src/config.mjs CHANGED
@@ -498,6 +498,9 @@ export function loadGitHubAuth(env, fileExists) {
498
498
  return { source, patVar, appId, installationId, privateKeyPath, privateKey };
499
499
  }
500
500
 
501
+ // env-internal TMPDIR, TEMP: the OS temp dir, read to place the default job, log, graph and settings
502
+ // paths below. Not a variable of this project's and not a deployment knob: PI_JOBS_DIR, PI_LOGS_DIR,
503
+ // PI_GRAPH_DIR and PI_SETTINGS_FILE are how an operator moves any of them, and .env.example says so.
501
504
  function defaultJobsDir() {
502
505
  // Under the OS temp dir by default. Holds only the read-only /job inputs (prompt + .pi/); the
503
506
  // workspace for a local job is the operator's own folder, not here.
package/src/doctor.mjs CHANGED
@@ -2072,6 +2072,9 @@ async function envSetupChecks(env, seams) {
2072
2072
  }
2073
2073
  }
2074
2074
 
2075
+ // env-internal PI_ENV_SETUP: unit configuration, deliberately never an .env key. The wrappers capture
2076
+ // it BEFORE they source ./.env so that nothing able to write that file can name a script they run
2077
+ // (REQ-DEPLOYMENT-BOOTSTRAP). doctor reads it here only to answer for a host whose unit names none.
2075
2078
  const fromEnv = (env.PI_ENV_SETUP ?? "").trim();
2076
2079
  if (sources.size === 0 && fromEnv) sources.set(fromEnv, "PI_ENV_SETUP in this environment");
2077
2080
 
@@ -194,9 +194,9 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
194
194
 
195
195
  /**
196
196
  * Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
197
- * `setTimeout` -- and `.unref()`'d so it can never hold the process open, which is the posture the
198
- * three `fs.watch` watchers already take. `stop` is registered as an extraCloser beside the runtime
199
- * queue, so a clean shutdown clears it before `process.exit`.
197
+ * `setTimeout` -- and `.unref()`'d so it can never hold the process open. `close` is registered as an
198
+ * extraCloser beside the runtime queue, so a clean shutdown clears it before `process.exit`; the three
199
+ * `fs.watch` watchers take the same two-part posture since issue #295, unref'd AND closed.
200
200
  */
201
201
  async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
202
202
  if (closed || timer) return; // a second start would leak the first interval
package/src/index.mjs CHANGED
@@ -836,10 +836,30 @@ export function createWorker({ connection, name, stopContainer, containerName, h
836
836
  // outlive the handler that was meant to stop them.
837
837
  for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
838
838
  for (const w of workers) await w.close().catch(() => {});
839
- // Close auxiliary resources (e.g. a cron scheduler) after the worker drains. Per-item catch
840
- // so one failing or absent closer never strands the others or blocks exit -- matches the
841
- // swallow posture on cancelAllJobs above.
842
- await Promise.all(extraClosers.map((c) => Promise.resolve(c.close?.()).catch(() => {})));
839
+ // Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
840
+ // Per-item catch so one failing or absent closer never strands the others or blocks exit -- matches
841
+ // the swallow posture on cancelAllJobs above. The try/catch is NOT redundant with the `.catch`:
842
+ // `Promise.resolve(x)` does not catch a SYNCHRONOUS throw from `x`, and `c.close` on a null entry
843
+ // throws before `Promise.resolve` is ever reached. Either would escape this callback, reject the whole
844
+ // shutdown and skip the `process.exit(0)` below. Jobs and containers are already stopped by then, so
845
+ // what a stranded loop leaks is the rest of the list: `registry.close()` is the DEL that keeps a
846
+ // stopped host from lingering as a ghost peer for its full TTL, and a ghost peer with a stale
847
+ // `fpCron` is what makes a later `reconcileGated` refuse a legitimate reconcile. The comment above
848
+ // promised this isolation before the code delivered it (issue #295). It bounds nothing, though: a
849
+ // closer that never settles still blocks exit, which no closer here does.
850
+ //
851
+ // Read LATE and deliberately: `start.mjs` pushes its live-edit watchers into this array AFTER handing
852
+ // it over, because they are armed after the boot reconcile. Anything here that snapshots or copies
853
+ // the array un-registers them in silence.
854
+ await Promise.all(
855
+ extraClosers.map((c) => {
856
+ try {
857
+ return Promise.resolve(c?.close?.()).catch(() => {});
858
+ } catch {
859
+ return Promise.resolve();
860
+ }
861
+ }),
862
+ );
843
863
  process.exit(0);
844
864
  };
845
865
  process.once("SIGTERM", shutdown);
@@ -140,6 +140,8 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
140
140
  workspace: resolved.manifest.workspace,
141
141
  jobDir: resolved.manifest.dir,
142
142
  publish,
143
+ // env-internal TERM: the operator's own terminal type, forwarded so the sandbox shell renders the
144
+ // way their terminal does. Nothing a deployment declares.
143
145
  term: env.TERM,
144
146
  idleSeconds: config.sandboxIdleMinutes * 60,
145
147
  network,
package/src/service.mjs CHANGED
@@ -265,6 +265,8 @@ export async function runService(argv = [], deps = {}) {
265
265
  moduleDir = MODULE_DIR,
266
266
  resolveReceiver = resolveReceiverStart,
267
267
  home = homedir(),
268
+ // env-internal USER: whose account a rendered unit runs as, taken from the login already running
269
+ // this command. An operator changes it by running the command as someone else, not by declaring it.
268
270
  user = env.USER || userInfo().username,
269
271
  tmp = tmpdir(),
270
272
  fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync },
package/src/start.mjs CHANGED
@@ -84,57 +84,144 @@ const WORKER_VERSION = (() => {
84
84
  * `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
85
85
  * `reaper_skipped`, and boot continues to the worker.
86
86
  */
87
+ /**
88
+ * The stop handle every live-edit watch below hands back, so `startWorker` can register it in the same
89
+ * `extraClosers` list that already closes the queues and the host registry (`index.mjs` -> shutdown).
90
+ *
91
+ * A WATCH NOTHING CAN CLOSE IS NOT A DETAIL (issue #295). `watch(dir, cb).unref?.()` retained nothing, so
92
+ * the watch outlived the worker that armed it, and the reload it later fired ran through THAT boot's
93
+ * `log` closure: that boot's injected `write`, stamped with that boot's `workerName`. One process running
94
+ * one worker, that is a rounding error at exit. One process running forty boots, which is what a test file
95
+ * is, and a worker that shut down two tests ago writes into a live worker's capture under a host that is
96
+ * not running -- `every log line carries the host` went red in CI reading `runnervmejwal` where it
97
+ * asserted `mac-mini-1`.
98
+ *
99
+ * UNREF'D IS NOT CLEANED UP, and that difference is what hid this across three features. `unref` says only
100
+ * that a handle will not hold the event loop open; the watch stays armed either way.
101
+ * `INT-HOST-REGISTRY-CONTRACT` states the same distinction from the opposite side, where a bound's timer
102
+ * is deliberately NOT unref'd because an unref'd timer does not fire when the hung command is the last
103
+ * thing holding the loop.
104
+ *
105
+ * Three properties, each one a way the shutdown breaks without it:
106
+ *
107
+ * - It CLOSES the FSWatcher, which is the leak itself.
108
+ * - It CANCELS the debounce the watcher already armed. Closing a watcher does not cancel a `setTimeout`
109
+ * the callback already set, and only the watcher was ever unref'd -- the 150ms timer never was. In a
110
+ * real worker that costs nothing, because the shutdown ends in `process.exit(0)` either way; it is the
111
+ * harness, where the loop is left to drain on its own, that the stray timer reaches.
112
+ * - Its `close()` NEVER THROWS for the handles its three callers build, and is idempotent -- by NULLING
113
+ * what it closed rather than by an early return, which would be a guard with nothing behind it. Node's
114
+ * own `FSWatcher.close()` is already both (measured: a second close returns early and neither throws),
115
+ * but this closer must not
116
+ * INHERIT that guarantee, it must MAKE it: the shutdown loop in `index.mjs` cannot catch a SYNCHRONOUS
117
+ * throw from a closer, and the comment there carries the argument. The swallow is
118
+ * `makeHostRegistry.close`'s posture rather than a new one.
119
+ *
120
+ * A watch that was never created -- the `catch` arm of each function below, a platform without `fs.watch`
121
+ * -- still gets a closer, so registration is unconditional and the list's shape never depends on the
122
+ * platform. That is why all three return from OUTSIDE their try/catch.
123
+ *
124
+ * WHAT IT CANNOT DO, because the list above would otherwise read as complete: cancel a reload that has
125
+ * ALREADY started. `reloadSchedules` is async and awaits a Valkey round trip, so a debounce that fired
126
+ * just before the close is still running after it -- and the watchers stay armed for the whole drain
127
+ * ahead of the closer loop, not merely 150ms. That reload cannot be recalled, so what is gated instead is
128
+ * its VOICE: `reloadLog` below goes quiet once `closed` is set, and every reload is handed that instead
129
+ * of the boot's own `log`, which is what the
130
+ * issue actually asks for -- a stopped worker writes no line. The reload's own Valkey work may still be
131
+ * cut off mid-flight by the queue closing beside it, leaving a scheduler set the next boot's reconcile
132
+ * repairs; that race predates this change and is not narrowed by it.
133
+ *
134
+ * EXPORTED for the reason `reloadScopedLimits` is: none of the three properties is observable through a
135
+ * real `fs.watch` without racing the filesystem, and a guarantee the shutdown rests on deserves a
136
+ * deterministic pin rather than a sleep.
137
+ */
138
+ export function makeWatchCloser(handles, log) {
139
+ return {
140
+ // The reload's voice, and the reason this factory is handed the boot's `log` rather than only its
141
+ // handles. A reload already in flight cannot be recalled, so what the close gates is what it can
142
+ // still SAY: after `closed`, a line from this watch would carry the host of a worker that has
143
+ // stopped, which is the bleed the issue is about. The arming lines keep the real `log` -- they run
144
+ // before any close.
145
+ reloadLog: (event, fields) => {
146
+ if (!handles.closed) log(event, fields);
147
+ },
148
+ close() {
149
+ // `closed` FIRST, before anything is torn down: it is what gates `reloadLog` above and the watch
150
+ // callback below, so a callback or a reload landing mid-close is already silenced.
151
+ handles.closed = true;
152
+ clearTimeout(handles.timer);
153
+ handles.timer = null;
154
+ try {
155
+ handles.watcher?.close();
156
+ } catch {
157
+ // A close that failed has already stopped mattering, and a THROW here rejects the shutdown.
158
+ }
159
+ handles.watcher = null;
160
+ },
161
+ };
162
+ }
163
+
87
164
  /**
88
165
  * Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
89
166
  * inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
90
- * `reloadSchedules`. Best-effort and unref'd so it never blocks shutdown; a platform without `fs.watch`
91
- * logs and the worker keeps its boot-time schedulers.
167
+ * `reloadSchedules`. Best-effort: a platform without `fs.watch` logs and the worker keeps its boot-time
168
+ * schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
169
+ * `startWorker` registers so the watch dies with the worker that armed it (issue #295).
92
170
  */
93
171
  function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
94
172
  const path = config.triggersFile;
95
173
  const dir = dirname(path) || ".";
96
174
  const file = basename(path);
97
- let timer = null;
175
+ const handles = { watcher: null, timer: null, closed: false };
176
+ const closer = makeWatchCloser(handles, log);
98
177
  try {
99
- watch(dir, (_event, changed) => {
178
+ handles.watcher = watch(dir, (_event, changed) => {
179
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
100
180
  if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
101
- clearTimeout(timer);
102
- timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
103
- }).unref?.();
181
+ clearTimeout(handles.timer);
182
+ handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), 150);
183
+ });
184
+ handles.watcher.unref?.();
104
185
  log("triggers_watching", { path });
105
186
  } catch (err) {
106
187
  log("triggers_watch_unavailable", { reason: err?.message });
107
188
  }
189
+ return closer;
108
190
  }
109
191
 
110
192
  /**
111
193
  * Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
112
194
  * and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
113
- * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort + unref'd.
195
+ * effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
196
+ * FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
114
197
  */
115
198
  function watchPauseWindowsFile(config, ref, log) {
116
199
  const path = config.pauseWindowsFile;
117
200
  const dir = dirname(path) || ".";
118
201
  const file = basename(path);
119
- let timer = null;
202
+ const handles = { watcher: null, timer: null, closed: false };
203
+ const closer = makeWatchCloser(handles, log);
120
204
  const reload = () => {
121
205
  try {
122
206
  ref.current = loadPauseWindows(config);
123
- log("pause_windows_reloaded", { count: ref.current.length });
207
+ closer.reloadLog("pause_windows_reloaded", { count: ref.current.length });
124
208
  } catch (err) {
125
- log("pause_windows_reload_invalid", { reason: err?.message });
209
+ closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
126
210
  }
127
211
  };
128
212
  try {
129
- watch(dir, (_event, changed) => {
213
+ handles.watcher = watch(dir, (_event, changed) => {
214
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
130
215
  if (changed && changed !== file) return;
131
- clearTimeout(timer);
132
- timer = setTimeout(reload, 150);
133
- }).unref?.();
216
+ clearTimeout(handles.timer);
217
+ handles.timer = setTimeout(reload, 150);
218
+ });
219
+ handles.watcher.unref?.();
134
220
  log("pause_windows_watching", { path });
135
221
  } catch (err) {
136
222
  log("pause_windows_watch_unavailable", { reason: err?.message });
137
223
  }
224
+ return closer;
138
225
  }
139
226
 
140
227
  /**
@@ -154,23 +241,28 @@ export function reloadScopedLimits(config, ref, log) {
154
241
 
155
242
  /**
156
243
  * Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
157
- * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort + unref'd.
244
+ * for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
245
+ * unref'd and the returned closer stops the watch with the worker (issue #295).
158
246
  */
159
247
  function watchScopedLimitsFile(config, ref, log) {
160
248
  const path = config.scopedLimitsFile;
161
249
  const dir = dirname(path) || ".";
162
250
  const file = basename(path);
163
- let timer = null;
251
+ const handles = { watcher: null, timer: null, closed: false };
252
+ const closer = makeWatchCloser(handles, log);
164
253
  try {
165
- watch(dir, (_event, changed) => {
254
+ handles.watcher = watch(dir, (_event, changed) => {
255
+ if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
166
256
  if (changed && changed !== file) return;
167
- clearTimeout(timer);
168
- timer = setTimeout(() => reloadScopedLimits(config, ref, log), 150);
169
- }).unref?.();
257
+ clearTimeout(handles.timer);
258
+ handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), 150);
259
+ });
260
+ handles.watcher.unref?.();
170
261
  log("scoped_limits_watching", { path });
171
262
  } catch (err) {
172
263
  log("scoped_limits_watch_unavailable", { reason: err?.message });
173
264
  }
265
+ return closer;
174
266
  }
175
267
 
176
268
  /**
@@ -677,6 +769,18 @@ export async function startWorker(
677
769
  reaps: backendReaps,
678
770
  });
679
771
 
772
+ // The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
773
+ // array rather than the literal it used to be, because up to three of its members do not exist yet: the
774
+ // live-edit watchers are armed at the END of boot, below, and deliberately after the boot reconcile --
775
+ // arming them earlier would let an operator edit run `reloadSchedules` concurrently with the boot
776
+ // `reconcileGated`, on a different queue handle, and reconcile's orphan prune is not safe against that.
777
+ //
778
+ // PUSHING AFTER THE HANDOFF IS SOUND FOR ONE REASON ONLY: `index.mjs` reads this array at SHUTDOWN time,
779
+ // not when it receives it, and so does the test harness at teardown. A refactor that COPIES it there --
780
+ // a spread, a freeze, a snapshot inside `createWorker` -- un-registers the watchers in SILENCE and puts
781
+ // issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
782
+ const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
783
+
680
784
  const worker = createWorkerFn({
681
785
  connection: parseConnection(config.valkeyUrl),
682
786
  // #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
@@ -696,7 +800,7 @@ export async function startWorker(
696
800
  getSettings,
697
801
  redis,
698
802
  recordRun,
699
- extraClosers: [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])],
803
+ extraClosers,
700
804
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
701
805
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
702
806
  pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
@@ -931,20 +1035,21 @@ export async function startWorker(
931
1035
 
932
1036
  // DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
933
1037
  // on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
934
- // Only when a triggers file is configured; best-effort + unref'd; a bad edit keeps the running schedulers.
1038
+ // Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
1039
+ // closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
935
1040
  if (config.triggersFile) {
936
- watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
1041
+ extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared));
937
1042
  }
938
1043
 
939
1044
  // REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
940
1045
  // an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
941
1046
  if (config.pauseWindowsFile) {
942
- watchPauseWindowsFile(config, pauseWindows, log);
1047
+ extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log));
943
1048
  }
944
1049
 
945
1050
  // Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
946
1051
  if (config.scopedLimitsFile) {
947
- watchScopedLimitsFile(config, scopedLimits, log);
1052
+ extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log));
948
1053
  }
949
1054
 
950
1055
  log("worker_started", {