@edgehero/pi-dispatch 1.10.1 → 1.10.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +80 -0
- package/package.json +1 -1
- package/src/config.mjs +3 -0
- package/src/doctor.mjs +3 -0
- package/src/host-registry.mjs +3 -3
- package/src/index.mjs +24 -4
- package/src/sandbox-cli.mjs +2 -0
- package/src/service.mjs +2 -0
- package/src/start.mjs +131 -26
package/.env.example
CHANGED
|
@@ -86,12 +86,22 @@ PI_JOB_IMAGE=pi-job:latest # the DEFAULT job image. Any trigger may nam
|
|
|
86
86
|
# PI_SUBSCRIPTIONS_FILE= # path to subscriptions.json — operator-declared subscription plan prices (the admin defaults to ./subscriptions.json in its working directory). Read by the ADMIN EXTENSION only, never at job time.
|
|
87
87
|
# Subscription-backed providers bill 0 per run (their rate tables are all zeros), so this file is where the real price lives — cost analytics only; it changes no routing, auth, or job behavior
|
|
88
88
|
# PI_SETTINGS_FILE= # ABSOLUTE path to the runtime settings overlay (default: OS temp /pi-dispatch/settings.json); edited by the admin extension, read by the worker per job
|
|
89
|
+
# PI_DISPATCH_DEPLOYMENT_FILE= # ABSOLUTE path to the deployment pointer the /dispatch panel reads to find a deployment built elsewhere
|
|
90
|
+
# (default: <your pi agent dir>/pi-dispatch-deployment.json). Read by the ADMIN EXTENSION only; the worker and receiver never look at it.
|
|
91
|
+
# Your own environment still wins key by key, so this points the panel at a deployment, it does not override one
|
|
92
|
+
# PI_GRAPH_DIR= # where `/dispatch insights` writes its HTML artifact (default: under the OS temp dir). Admin extension only. See docs/insights.md
|
|
89
93
|
|
|
90
94
|
# --- Reuse your existing pi setup in every job (see docs/global-pi-overlay.md) ---
|
|
91
95
|
# PI_GLOBAL_PI_DIR= # dir with your host pi setup (models.json/skills/APPEND_SYSTEM.md), mounted /opt/pi-global:ro into every job, layered UNDER each repo's .pi/. Unset = off. Stage it with: pi-dispatch import-pi
|
|
92
96
|
# PI_GLOBAL_ALLOW_EXTENSIONS= # the overlay's extensions LOAD by default (staging them with import-pi, which prints each one, is the vetting step). Set exactly 0 to keep them staged but dormant.
|
|
93
97
|
# Unset, empty and the legacy 1 all mean LOAD. ANY other value refuses to boot -- a typo must never silently leave code running against adversarial input with open egress.
|
|
94
98
|
# This knob covers the OVERLAY only. A serviced repo's own /workspace/.pi/extensions load regardless (they are default-branch, merge-gated) -- see SECURITY.md.
|
|
99
|
+
# PI_CODING_AGENT_DIR= # where YOUR pi setup lives on this host (default: ~/.pi/agent). `pi-dispatch import-pi` reads its models.json,
|
|
100
|
+
# skills and APPEND_SYSTEM.md from here, and the panel looks here for the deployment pointer.
|
|
101
|
+
# It is also read AT JOB TIME: with no provider key in the environment the worker reads this directory's auth.json
|
|
102
|
+
# for one (on by default; PI_AUTH_FROM_PI=0 turns it off), so pointing this at the wrong place makes every job
|
|
103
|
+
# refuse pre-spend with no credential for the provider. It is the SOURCE that gets staged, never the thing mounted:
|
|
104
|
+
# PI_GLOBAL_PI_DIR above is what a job actually sees. Set it if your pi lives somewhere other than your home directory
|
|
95
105
|
# PI_PACKAGES_FILE= # path to pi-packages.json (default: ./pi-packages.json; --packages-file wins). Read ONLY by `pi-dispatch import-pi --with-packages`, never at job time.
|
|
96
106
|
# Staged packages/ rides INSIDE PI_GLOBAL_PI_DIR -- no separate mount, no separate env dir -- and loads for EVERY job once staged; decline it PER TRIGGER with "packages": false in triggers.json, NOT by an env flag.
|
|
97
107
|
# Versions must be EXACT (no ^ ~ * or latest); staging uses --ignore-scripts, so a package needing a build step is staged INCOMPLETE and import-pi warns.
|
|
@@ -166,6 +176,12 @@ RECEIVER_BIND=0.0.0.0
|
|
|
166
176
|
GITHUB_AUTH_SOURCE=gh
|
|
167
177
|
# For GITHUB_AUTH_SOURCE=pat: a repo-scoped, short-expiry fine-grained PAT
|
|
168
178
|
GITHUB_PAT=
|
|
179
|
+
# GITHUB_PAT_VAR= # which variable above actually holds the PAT. Default GITHUB_PAT; set this only if your
|
|
180
|
+
# secrets manager insists on its own name and you would rather not copy the value to a second key.
|
|
181
|
+
# The NAME is not checked against anything: whatever you put here is read verbatim, so a typo
|
|
182
|
+
# reads an empty variable and the worker refuses at boot naming the name you chose. Pointing it
|
|
183
|
+
# at a variable that holds something else (GITHUB_APP_PRIVATE_KEY, say) is the mistake worth
|
|
184
|
+
# knowing about, because nothing stops it and the PAT path would then send that value to GitHub.
|
|
169
185
|
# For GITHUB_AUTH_SOURCE=app (optional; required for multi-tenant)
|
|
170
186
|
# `pi-dispatch setup github` fills all three in one browser click (App Manifest flow) and writes the PEM 0600
|
|
171
187
|
GITHUB_APP_ID=
|
|
@@ -176,6 +192,23 @@ GITHUB_APP_PRIVATE_KEY_PATH=
|
|
|
176
192
|
# escapes both work. Never list it in PI_FORWARD_ENV: it mints tokens for every repo the App is on.
|
|
177
193
|
GITHUB_APP_PRIVATE_KEY=
|
|
178
194
|
|
|
195
|
+
# --- Polling ingest, instead of a webhook (GitHub only) --- issue #282
|
|
196
|
+
# `pi-dispatch-receiver poll` fetches issue events, comments and pull requests over TLS with your own
|
|
197
|
+
# credential, so a deployment with no public URL, no DNS and no tunnel still fires triggers. Same gates and
|
|
198
|
+
# same queue as the webhook path; about a minute of latency instead of a second. Nothing here is read by
|
|
199
|
+
# `pi-dispatch-receiver serve`, and no WEBHOOK_SECRET is needed to poll: there is no inbound delivery to
|
|
200
|
+
# verify, because the poller originates every request itself. See docs/polling.md.
|
|
201
|
+
# POLL_REPOS= # WHICH repos to watch: comma-separated owner/name (e.g. acme/web,acme/api). Duplicates are dropped.
|
|
202
|
+
# Each entry must be exactly owner/name -- one slash, no spaces -- or the receiver refuses at boot naming the bad entry.
|
|
203
|
+
# Leave it UNSET only with GITHUB_AUTH_SOURCE=app: the poller then lists the App installation's own repos and
|
|
204
|
+
# re-lists every tenth cycle, so installing the App on a new repo starts polling it without an edit here.
|
|
205
|
+
# Unset under any other auth source is a boot refusal, deliberately: a PAT names no repo set, and a poller
|
|
206
|
+
# watching nothing looks exactly like a poller that is working.
|
|
207
|
+
# POLL_INTERVAL_SECONDS= # seconds between cycles. Default 60, floored at 30: a positive value BELOW the floor is raised to it,
|
|
208
|
+
# while 0, a negative, a fraction or junk still refuses at boot.
|
|
209
|
+
# GitHub asks pollers to respect its own x-poll-interval hint, which is honored as a MINIMUM when it arrives,
|
|
210
|
+
# so a busy hour slows the loop down rather than the loop hammering the API. A typo'd 1 must not turn this into a hammer.
|
|
211
|
+
|
|
179
212
|
# --- GitLab trigger (receiver + worker auth) ---
|
|
180
213
|
# Optional. Set these only to service GitLab projects; leaving GITLAB_TOKEN unset means no /gitlab
|
|
181
214
|
# endpoint exists at all, rather than one that answers 401. See docs/gitlab.md.
|
|
@@ -184,6 +217,13 @@ GITHUB_APP_PRIVATE_KEY=
|
|
|
184
217
|
# scope that can post a note -- GitLab offers no contents-vs-issues split -- so scope it to one project
|
|
185
218
|
# and rotate it (CONST-TOKEN-SCOPED-PER-JOB). A GROUP token reaches every project in the group.
|
|
186
219
|
GITLAB_TOKEN=
|
|
220
|
+
# GITLAB_AUTH_SOURCE= # accepts exactly one value, "pat", which is also the default, so there is nothing to set here.
|
|
221
|
+
# It exists to REFUSE the wrong assumption rather than to offer a choice: GITHUB_AUTH_SOURCE has
|
|
222
|
+
# three sources, and an operator who reasons by symmetry and writes app here gets a sentence at
|
|
223
|
+
# boot saying GitLab has no App equivalent, instead of a knob that is silently ignored.
|
|
224
|
+
# The refusal needs GITLAB_TOKEN to be set: with no token there is no GitLab to configure, the
|
|
225
|
+
# whole block is skipped, and a stray app here really is ignored. Same for the two below.
|
|
226
|
+
# FORGEJO_AUTH_SOURCE and AZURE_AUTH_SOURCE are the same variable for the same reason.
|
|
187
227
|
# Your instance root. Only for self-hosted GitLab.
|
|
188
228
|
GITLAB_URL=https://gitlab.com
|
|
189
229
|
# How the receiver verifies a delivery. REQUIRED once any GITLAB_* variable is set, and deliberately not
|
|
@@ -205,6 +245,7 @@ FORGEJO_URL=
|
|
|
205
245
|
# expire: there is no App or installation token, so rotation is the whole mitigation
|
|
206
246
|
# (CONST-TOKEN-SCOPED-PER-JOB).
|
|
207
247
|
FORGEJO_TOKEN=
|
|
248
|
+
# FORGEJO_AUTH_SOURCE= # only "pat", the default. See GITLAB_AUTH_SOURCE above for why it exists at all.
|
|
208
249
|
# The harness account's NUMERIC id. Required when the token above is repository-scoped, because such a
|
|
209
250
|
# token may not carry read:user and therefore cannot call GET /user. The receiver refuses to boot without an
|
|
210
251
|
# identity from one source or the other: the bot-loop guard compares against it, and an unresolved identity
|
|
@@ -223,6 +264,7 @@ AZURE_ORG_URL=
|
|
|
223
264
|
# permissions in Project Settings -- not from the token's scopes. It also needs vso.graph, to resolve the
|
|
224
265
|
# actor's project membership before a job may be enqueued.
|
|
225
266
|
AZURE_TOKEN=
|
|
267
|
+
# AZURE_AUTH_SOURCE= # only "pat", the default. See GITLAB_AUTH_SOURCE above for why it exists at all.
|
|
226
268
|
# REQUIRED once any AZURE_* variable is set, and deliberately not defaulted: both modes are shared-secret
|
|
227
269
|
# compares that cover no bytes, so which header carries the secret must be a choice somebody made.
|
|
228
270
|
# basic -- Authorization: Basic <base64>, the credential you set on the subscription
|
|
@@ -232,3 +274,41 @@ AZURE_WEBHOOK_MODE=
|
|
|
232
274
|
AZURE_WEBHOOK_SECRET=
|
|
233
275
|
# Required only when AZURE_WEBHOOK_MODE=header.
|
|
234
276
|
AZURE_WEBHOOK_HEADER=
|
|
277
|
+
|
|
278
|
+
# --- What is deliberately NOT a key in this file --- issue #282
|
|
279
|
+
# The rule, and it is checked by a test (worker/test/env-docs.test.mjs): every environment variable this
|
|
280
|
+
# project's loaders read is either a key above, or named here with the reason it cannot be one. A variable
|
|
281
|
+
# that is read and appears in neither is the failure this section exists to prevent, because an operator has
|
|
282
|
+
# no way to discover it and no way to find out that setting it did nothing.
|
|
283
|
+
#
|
|
284
|
+
# The worker's own inputs to a job container: PI_JOB_ID, PI_FLOW, PI_COMMAND, PI_PACKAGES, PI_SESSION_FILE,
|
|
285
|
+
# PI_OFFLINE, and the three PLAYWRIGHT_ names. The container's environment is BUILT, not inherited: the worker
|
|
286
|
+
# passes exactly the map it composed, so a value set here never reaches a job to be overridden in the first
|
|
287
|
+
# place. Several of them are also conditional, present only when the job has a flow, a command, staged
|
|
288
|
+
# packages or a session to resume. These are the container's side of INT-CONTAINER-RUNTIME-CONTRACT
|
|
289
|
+
# (specs/interfaces.md, docs/job-image.md), and run.secrets refuses the names at load so a trigger cannot
|
|
290
|
+
# bind one either. PI_FORWARD_ENV is NOT checked against these names, and it is applied after the worker's
|
|
291
|
+
# own map, so naming one there does override it. Not everything is: run.secrets, the egress variables and
|
|
292
|
+
# the minted forge token are all written later still and win over a forwarded value. A real edge, not a
|
|
293
|
+
# recommendation.
|
|
294
|
+
#
|
|
295
|
+
# PI_ENV_SETUP is an argument to `pi-dispatch service render|install --env-setup <absolute path>`, never a
|
|
296
|
+
# key. The service wrappers capture it BEFORE they source ./.env, precisely so that anything able to write
|
|
297
|
+
# this file cannot name a script the wrapper will run as the worker. A line here is honored by nothing, and
|
|
298
|
+
# that is the point rather than an oversight (docs/secrets.md, REQ-DEPLOYMENT-BOOTSTRAP).
|
|
299
|
+
#
|
|
300
|
+
# PI_RETRY_MAX (default 2) and PI_RETRY_BASE_MS (default 2000) are read by the runner INSIDE the container,
|
|
301
|
+
# and nothing on the host writes them, so a line here sets them on this machine and never reaches a job.
|
|
302
|
+
# PI_FORWARD_ENV is what carries a host value into a container, and is how you would actually change them.
|
|
303
|
+
#
|
|
304
|
+
# Provider key names are pi's, not ours. The worker asks pi which variable your PI_PROVIDER expects, so the
|
|
305
|
+
# provider block at the top of this file lists the common ones as examples and is not the whole set: pi
|
|
306
|
+
# supports around thirty providers and the answer travels with pi rather than with this file.
|
|
307
|
+
#
|
|
308
|
+
# Read from the surrounding system, not from a deployment: TMPDIR and TEMP (where the default job, log,
|
|
309
|
+
# graph and settings paths go), USER (the account a rendered service unit runs as), and TERM, SSH_CONNECTION,
|
|
310
|
+
# SSH_TTY, DISPLAY, WAYLAND_DISPLAY (how the panel decides whether it can open a browser for you).
|
|
311
|
+
#
|
|
312
|
+
# PI_DISPATCH_REQUIRE_LOADER_TESTS, PI_DISPATCH_REQUIRE_WORKER_TESTS, PI_DISPATCH_REQUIRE_RECEIVER_TESTS and
|
|
313
|
+
# VALKEY_TEST_URL turn locally skipped integration tests into required ones. They are read by the test files
|
|
314
|
+
# themselves and by CI, never by the worker, the receiver or the panel.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "1.10.
|
|
3
|
+
"version": "1.10.3",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Self-hosted job harness for the pi coding agent: a BullMQ worker that drains the queue, mints scoped forge tokens, and runs one container per job — plus the pi-dispatch CLI (init, up, doctor, service).",
|
|
6
6
|
"keywords": [
|
package/src/config.mjs
CHANGED
|
@@ -498,6 +498,9 @@ export function loadGitHubAuth(env, fileExists) {
|
|
|
498
498
|
return { source, patVar, appId, installationId, privateKeyPath, privateKey };
|
|
499
499
|
}
|
|
500
500
|
|
|
501
|
+
// env-internal TMPDIR, TEMP: the OS temp dir, read to place the default job, log, graph and settings
|
|
502
|
+
// paths below. Not a variable of this project's and not a deployment knob: PI_JOBS_DIR, PI_LOGS_DIR,
|
|
503
|
+
// PI_GRAPH_DIR and PI_SETTINGS_FILE are how an operator moves any of them, and .env.example says so.
|
|
501
504
|
function defaultJobsDir() {
|
|
502
505
|
// Under the OS temp dir by default. Holds only the read-only /job inputs (prompt + .pi/); the
|
|
503
506
|
// workspace for a local job is the operator's own folder, not here.
|
package/src/doctor.mjs
CHANGED
|
@@ -2072,6 +2072,9 @@ async function envSetupChecks(env, seams) {
|
|
|
2072
2072
|
}
|
|
2073
2073
|
}
|
|
2074
2074
|
|
|
2075
|
+
// env-internal PI_ENV_SETUP: unit configuration, deliberately never an .env key. The wrappers capture
|
|
2076
|
+
// it BEFORE they source ./.env so that nothing able to write that file can name a script they run
|
|
2077
|
+
// (REQ-DEPLOYMENT-BOOTSTRAP). doctor reads it here only to answer for a host whose unit names none.
|
|
2075
2078
|
const fromEnv = (env.PI_ENV_SETUP ?? "").trim();
|
|
2076
2079
|
if (sources.size === 0 && fromEnv) sources.set(fromEnv, "PI_ENV_SETUP in this environment");
|
|
2077
2080
|
|
package/src/host-registry.mjs
CHANGED
|
@@ -194,9 +194,9 @@ export function makeHostRegistry({ redis, name, now = () => Date.now(), ttlMs =
|
|
|
194
194
|
|
|
195
195
|
/**
|
|
196
196
|
* Start beating. ONE `setInterval` -- the first in `worker/src`, every other timer here being a
|
|
197
|
-
* `setTimeout` -- and `.unref()`'d so it can never hold the process open
|
|
198
|
-
*
|
|
199
|
-
*
|
|
197
|
+
* `setTimeout` -- and `.unref()`'d so it can never hold the process open. `close` is registered as an
|
|
198
|
+
* extraCloser beside the runtime queue, so a clean shutdown clears it before `process.exit`; the three
|
|
199
|
+
* `fs.watch` watchers take the same two-part posture since issue #295, unref'd AND closed.
|
|
200
200
|
*/
|
|
201
201
|
async start(fields = {}, { intervalMs = HOST_BEAT_MS } = {}) {
|
|
202
202
|
if (closed || timer) return; // a second start would leak the first interval
|
package/src/index.mjs
CHANGED
|
@@ -836,10 +836,30 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
836
836
|
// outlive the handler that was meant to stop them.
|
|
837
837
|
for (const w of workers) await Promise.resolve(w.cancelAllJobs?.("shutdown")).catch(() => {});
|
|
838
838
|
for (const w of workers) await w.close().catch(() => {});
|
|
839
|
-
// Close auxiliary resources (
|
|
840
|
-
// so one failing or absent closer never strands the others or blocks exit -- matches
|
|
841
|
-
// swallow posture on cancelAllJobs above.
|
|
842
|
-
|
|
839
|
+
// Close auxiliary resources (a cron scheduler, the live-edit file watchers) after the worker drains.
|
|
840
|
+
// Per-item catch so one failing or absent closer never strands the others or blocks exit -- matches
|
|
841
|
+
// the swallow posture on cancelAllJobs above. The try/catch is NOT redundant with the `.catch`:
|
|
842
|
+
// `Promise.resolve(x)` does not catch a SYNCHRONOUS throw from `x`, and `c.close` on a null entry
|
|
843
|
+
// throws before `Promise.resolve` is ever reached. Either would escape this callback, reject the whole
|
|
844
|
+
// shutdown and skip the `process.exit(0)` below. Jobs and containers are already stopped by then, so
|
|
845
|
+
// what a stranded loop leaks is the rest of the list: `registry.close()` is the DEL that keeps a
|
|
846
|
+
// stopped host from lingering as a ghost peer for its full TTL, and a ghost peer with a stale
|
|
847
|
+
// `fpCron` is what makes a later `reconcileGated` refuse a legitimate reconcile. The comment above
|
|
848
|
+
// promised this isolation before the code delivered it (issue #295). It bounds nothing, though: a
|
|
849
|
+
// closer that never settles still blocks exit, which no closer here does.
|
|
850
|
+
//
|
|
851
|
+
// Read LATE and deliberately: `start.mjs` pushes its live-edit watchers into this array AFTER handing
|
|
852
|
+
// it over, because they are armed after the boot reconcile. Anything here that snapshots or copies
|
|
853
|
+
// the array un-registers them in silence.
|
|
854
|
+
await Promise.all(
|
|
855
|
+
extraClosers.map((c) => {
|
|
856
|
+
try {
|
|
857
|
+
return Promise.resolve(c?.close?.()).catch(() => {});
|
|
858
|
+
} catch {
|
|
859
|
+
return Promise.resolve();
|
|
860
|
+
}
|
|
861
|
+
}),
|
|
862
|
+
);
|
|
843
863
|
process.exit(0);
|
|
844
864
|
};
|
|
845
865
|
process.once("SIGTERM", shutdown);
|
package/src/sandbox-cli.mjs
CHANGED
|
@@ -140,6 +140,8 @@ export async function runSandbox(argv = [], { env = process.env, deps = {} } = {
|
|
|
140
140
|
workspace: resolved.manifest.workspace,
|
|
141
141
|
jobDir: resolved.manifest.dir,
|
|
142
142
|
publish,
|
|
143
|
+
// env-internal TERM: the operator's own terminal type, forwarded so the sandbox shell renders the
|
|
144
|
+
// way their terminal does. Nothing a deployment declares.
|
|
143
145
|
term: env.TERM,
|
|
144
146
|
idleSeconds: config.sandboxIdleMinutes * 60,
|
|
145
147
|
network,
|
package/src/service.mjs
CHANGED
|
@@ -265,6 +265,8 @@ export async function runService(argv = [], deps = {}) {
|
|
|
265
265
|
moduleDir = MODULE_DIR,
|
|
266
266
|
resolveReceiver = resolveReceiverStart,
|
|
267
267
|
home = homedir(),
|
|
268
|
+
// env-internal USER: whose account a rendered unit runs as, taken from the login already running
|
|
269
|
+
// this command. An operator changes it by running the command as someone else, not by declaring it.
|
|
268
270
|
user = env.USER || userInfo().username,
|
|
269
271
|
tmp = tmpdir(),
|
|
270
272
|
fs = { existsSync, mkdirSync, readFileSync, unlinkSync, writeFileSync },
|
package/src/start.mjs
CHANGED
|
@@ -84,57 +84,144 @@ const WORKER_VERSION = (() => {
|
|
|
84
84
|
* `reap()` NEVER throws: a missing docker binary or a down daemon is caught, logged as
|
|
85
85
|
* `reaper_skipped`, and boot continues to the worker.
|
|
86
86
|
*/
|
|
87
|
+
/**
|
|
88
|
+
* The stop handle every live-edit watch below hands back, so `startWorker` can register it in the same
|
|
89
|
+
* `extraClosers` list that already closes the queues and the host registry (`index.mjs` -> shutdown).
|
|
90
|
+
*
|
|
91
|
+
* A WATCH NOTHING CAN CLOSE IS NOT A DETAIL (issue #295). `watch(dir, cb).unref?.()` retained nothing, so
|
|
92
|
+
* the watch outlived the worker that armed it, and the reload it later fired ran through THAT boot's
|
|
93
|
+
* `log` closure: that boot's injected `write`, stamped with that boot's `workerName`. One process running
|
|
94
|
+
* one worker, that is a rounding error at exit. One process running forty boots, which is what a test file
|
|
95
|
+
* is, and a worker that shut down two tests ago writes into a live worker's capture under a host that is
|
|
96
|
+
* not running -- `every log line carries the host` went red in CI reading `runnervmejwal` where it
|
|
97
|
+
* asserted `mac-mini-1`.
|
|
98
|
+
*
|
|
99
|
+
* UNREF'D IS NOT CLEANED UP, and that difference is what hid this across three features. `unref` says only
|
|
100
|
+
* that a handle will not hold the event loop open; the watch stays armed either way.
|
|
101
|
+
* `INT-HOST-REGISTRY-CONTRACT` states the same distinction from the opposite side, where a bound's timer
|
|
102
|
+
* is deliberately NOT unref'd because an unref'd timer does not fire when the hung command is the last
|
|
103
|
+
* thing holding the loop.
|
|
104
|
+
*
|
|
105
|
+
* Three properties, each one a way the shutdown breaks without it:
|
|
106
|
+
*
|
|
107
|
+
* - It CLOSES the FSWatcher, which is the leak itself.
|
|
108
|
+
* - It CANCELS the debounce the watcher already armed. Closing a watcher does not cancel a `setTimeout`
|
|
109
|
+
* the callback already set, and only the watcher was ever unref'd -- the 150ms timer never was. In a
|
|
110
|
+
* real worker that costs nothing, because the shutdown ends in `process.exit(0)` either way; it is the
|
|
111
|
+
* harness, where the loop is left to drain on its own, that the stray timer reaches.
|
|
112
|
+
* - Its `close()` NEVER THROWS for the handles its three callers build, and is idempotent -- by NULLING
|
|
113
|
+
* what it closed rather than by an early return, which would be a guard with nothing behind it. Node's
|
|
114
|
+
* own `FSWatcher.close()` is already both (measured: a second close returns early and neither throws),
|
|
115
|
+
* but this closer must not
|
|
116
|
+
* INHERIT that guarantee, it must MAKE it: the shutdown loop in `index.mjs` cannot catch a SYNCHRONOUS
|
|
117
|
+
* throw from a closer, and the comment there carries the argument. The swallow is
|
|
118
|
+
* `makeHostRegistry.close`'s posture rather than a new one.
|
|
119
|
+
*
|
|
120
|
+
* A watch that was never created -- the `catch` arm of each function below, a platform without `fs.watch`
|
|
121
|
+
* -- still gets a closer, so registration is unconditional and the list's shape never depends on the
|
|
122
|
+
* platform. That is why all three return from OUTSIDE their try/catch.
|
|
123
|
+
*
|
|
124
|
+
* WHAT IT CANNOT DO, because the list above would otherwise read as complete: cancel a reload that has
|
|
125
|
+
* ALREADY started. `reloadSchedules` is async and awaits a Valkey round trip, so a debounce that fired
|
|
126
|
+
* just before the close is still running after it -- and the watchers stay armed for the whole drain
|
|
127
|
+
* ahead of the closer loop, not merely 150ms. That reload cannot be recalled, so what is gated instead is
|
|
128
|
+
* its VOICE: `reloadLog` below goes quiet once `closed` is set, and every reload is handed that instead
|
|
129
|
+
* of the boot's own `log`, which is what the
|
|
130
|
+
* issue actually asks for -- a stopped worker writes no line. The reload's own Valkey work may still be
|
|
131
|
+
* cut off mid-flight by the queue closing beside it, leaving a scheduler set the next boot's reconcile
|
|
132
|
+
* repairs; that race predates this change and is not narrowed by it.
|
|
133
|
+
*
|
|
134
|
+
* EXPORTED for the reason `reloadScopedLimits` is: none of the three properties is observable through a
|
|
135
|
+
* real `fs.watch` without racing the filesystem, and a guarantee the shutdown rests on deserves a
|
|
136
|
+
* deterministic pin rather than a sleep.
|
|
137
|
+
*/
|
|
138
|
+
export function makeWatchCloser(handles, log) {
|
|
139
|
+
return {
|
|
140
|
+
// The reload's voice, and the reason this factory is handed the boot's `log` rather than only its
|
|
141
|
+
// handles. A reload already in flight cannot be recalled, so what the close gates is what it can
|
|
142
|
+
// still SAY: after `closed`, a line from this watch would carry the host of a worker that has
|
|
143
|
+
// stopped, which is the bleed the issue is about. The arming lines keep the real `log` -- they run
|
|
144
|
+
// before any close.
|
|
145
|
+
reloadLog: (event, fields) => {
|
|
146
|
+
if (!handles.closed) log(event, fields);
|
|
147
|
+
},
|
|
148
|
+
close() {
|
|
149
|
+
// `closed` FIRST, before anything is torn down: it is what gates `reloadLog` above and the watch
|
|
150
|
+
// callback below, so a callback or a reload landing mid-close is already silenced.
|
|
151
|
+
handles.closed = true;
|
|
152
|
+
clearTimeout(handles.timer);
|
|
153
|
+
handles.timer = null;
|
|
154
|
+
try {
|
|
155
|
+
handles.watcher?.close();
|
|
156
|
+
} catch {
|
|
157
|
+
// A close that failed has already stopped mattering, and a THROW here rejects the shutdown.
|
|
158
|
+
}
|
|
159
|
+
handles.watcher = null;
|
|
160
|
+
},
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
87
164
|
/**
|
|
88
165
|
* Watch the DIRECTORY holding the triggers file (robust to the admin's atomic tmp+rename, which swaps the
|
|
89
166
|
* inode a file-watch would lose), debounce, and re-reconcile the cron schedulers on change via
|
|
90
|
-
* `reloadSchedules`. Best-effort
|
|
91
|
-
*
|
|
167
|
+
* `reloadSchedules`. Best-effort: a platform without `fs.watch` logs and the worker keeps its boot-time
|
|
168
|
+
* schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
|
|
169
|
+
* `startWorker` registers so the watch dies with the worker that armed it (issue #295).
|
|
92
170
|
*/
|
|
93
171
|
function watchTriggersFile(config, queue, log, ref, registry, tz, fleet) {
|
|
94
172
|
const path = config.triggersFile;
|
|
95
173
|
const dir = dirname(path) || ".";
|
|
96
174
|
const file = basename(path);
|
|
97
|
-
|
|
175
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
176
|
+
const closer = makeWatchCloser(handles, log);
|
|
98
177
|
try {
|
|
99
|
-
watch(dir, (_event, changed) => {
|
|
178
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
179
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
100
180
|
if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
|
|
101
|
-
clearTimeout(timer);
|
|
102
|
-
timer = setTimeout(() => void reloadSchedules(config, queue, { log, ref, registry, tz, fleet }), 150);
|
|
103
|
-
})
|
|
181
|
+
clearTimeout(handles.timer);
|
|
182
|
+
handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }), 150);
|
|
183
|
+
});
|
|
184
|
+
handles.watcher.unref?.();
|
|
104
185
|
log("triggers_watching", { path });
|
|
105
186
|
} catch (err) {
|
|
106
187
|
log("triggers_watch_unavailable", { reason: err?.message });
|
|
107
188
|
}
|
|
189
|
+
return closer;
|
|
108
190
|
}
|
|
109
191
|
|
|
110
192
|
/**
|
|
111
193
|
* Watch the DIRECTORY holding the pause-windows file (same atomic-rename robustness as the triggers watch)
|
|
112
194
|
* and hot-swap the in-memory windows in `ref.current` on change. A bad edit keeps the last-good windows in
|
|
113
|
-
* effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort
|
|
195
|
+
* effect (OQ-008 live-edit safety) — the pause gate never loses its config to a typo. Best-effort; the
|
|
196
|
+
* FSWatcher is unref'd and the returned closer stops the watch with the worker (issue #295).
|
|
114
197
|
*/
|
|
115
198
|
function watchPauseWindowsFile(config, ref, log) {
|
|
116
199
|
const path = config.pauseWindowsFile;
|
|
117
200
|
const dir = dirname(path) || ".";
|
|
118
201
|
const file = basename(path);
|
|
119
|
-
|
|
202
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
203
|
+
const closer = makeWatchCloser(handles, log);
|
|
120
204
|
const reload = () => {
|
|
121
205
|
try {
|
|
122
206
|
ref.current = loadPauseWindows(config);
|
|
123
|
-
|
|
207
|
+
closer.reloadLog("pause_windows_reloaded", { count: ref.current.length });
|
|
124
208
|
} catch (err) {
|
|
125
|
-
|
|
209
|
+
closer.reloadLog("pause_windows_reload_invalid", { reason: err?.message });
|
|
126
210
|
}
|
|
127
211
|
};
|
|
128
212
|
try {
|
|
129
|
-
watch(dir, (_event, changed) => {
|
|
213
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
214
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
130
215
|
if (changed && changed !== file) return;
|
|
131
|
-
clearTimeout(timer);
|
|
132
|
-
timer = setTimeout(reload, 150);
|
|
133
|
-
})
|
|
216
|
+
clearTimeout(handles.timer);
|
|
217
|
+
handles.timer = setTimeout(reload, 150);
|
|
218
|
+
});
|
|
219
|
+
handles.watcher.unref?.();
|
|
134
220
|
log("pause_windows_watching", { path });
|
|
135
221
|
} catch (err) {
|
|
136
222
|
log("pause_windows_watch_unavailable", { reason: err?.message });
|
|
137
223
|
}
|
|
224
|
+
return closer;
|
|
138
225
|
}
|
|
139
226
|
|
|
140
227
|
/**
|
|
@@ -154,23 +241,28 @@ export function reloadScopedLimits(config, ref, log) {
|
|
|
154
241
|
|
|
155
242
|
/**
|
|
156
243
|
* Watch the scoped-limits file (issue #242) the way the pause-windows watcher above does: the DIRECTORY,
|
|
157
|
-
* for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort
|
|
244
|
+
* for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
|
|
245
|
+
* unref'd and the returned closer stops the watch with the worker (issue #295).
|
|
158
246
|
*/
|
|
159
247
|
function watchScopedLimitsFile(config, ref, log) {
|
|
160
248
|
const path = config.scopedLimitsFile;
|
|
161
249
|
const dir = dirname(path) || ".";
|
|
162
250
|
const file = basename(path);
|
|
163
|
-
|
|
251
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
252
|
+
const closer = makeWatchCloser(handles, log);
|
|
164
253
|
try {
|
|
165
|
-
watch(dir, (_event, changed) => {
|
|
254
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
255
|
+
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
166
256
|
if (changed && changed !== file) return;
|
|
167
|
-
clearTimeout(timer);
|
|
168
|
-
timer = setTimeout(() => reloadScopedLimits(config, ref,
|
|
169
|
-
})
|
|
257
|
+
clearTimeout(handles.timer);
|
|
258
|
+
handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), 150);
|
|
259
|
+
});
|
|
260
|
+
handles.watcher.unref?.();
|
|
170
261
|
log("scoped_limits_watching", { path });
|
|
171
262
|
} catch (err) {
|
|
172
263
|
log("scoped_limits_watch_unavailable", { reason: err?.message });
|
|
173
264
|
}
|
|
265
|
+
return closer;
|
|
174
266
|
}
|
|
175
267
|
|
|
176
268
|
/**
|
|
@@ -677,6 +769,18 @@ export async function startWorker(
|
|
|
677
769
|
reaps: backendReaps,
|
|
678
770
|
});
|
|
679
771
|
|
|
772
|
+
// The auxiliary handles the shutdown closes after the worker drains (`index.mjs` -> shutdown). A NAMED
|
|
773
|
+
// array rather than the literal it used to be, because up to three of its members do not exist yet: the
|
|
774
|
+
// live-edit watchers are armed at the END of boot, below, and deliberately after the boot reconcile --
|
|
775
|
+
// arming them earlier would let an operator edit run `reloadSchedules` concurrently with the boot
|
|
776
|
+
// `reconcileGated`, on a different queue handle, and reconcile's orphan prune is not safe against that.
|
|
777
|
+
//
|
|
778
|
+
// PUSHING AFTER THE HANDOFF IS SOUND FOR ONE REASON ONLY: `index.mjs` reads this array at SHUTDOWN time,
|
|
779
|
+
// not when it receives it, and so does the test harness at teardown. A refactor that COPIES it there --
|
|
780
|
+
// a spread, a freeze, a snapshot inside `createWorker` -- un-registers the watchers in SILENCE and puts
|
|
781
|
+
// issue #295 back. Append only: two tests pin `[0]` as the runtime queue and `[1]` as the registry.
|
|
782
|
+
const extraClosers = [runtimeQueue, registry, ...(cronQueue === runtimeQueue ? [] : [cronQueue])];
|
|
783
|
+
|
|
680
784
|
const worker = createWorkerFn({
|
|
681
785
|
connection: parseConnection(config.valkeyUrl),
|
|
682
786
|
// #227. The abort path's stop, resolved per job rather than hard-wired to docker. A container NAME is
|
|
@@ -696,7 +800,7 @@ export async function startWorker(
|
|
|
696
800
|
getSettings,
|
|
697
801
|
redis,
|
|
698
802
|
recordRun,
|
|
699
|
-
extraClosers
|
|
803
|
+
extraClosers,
|
|
700
804
|
// REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
|
|
701
805
|
// Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
|
|
702
806
|
pauseUntil: (job, now) => pauseUntilMs(pauseWindows.current, job, now),
|
|
@@ -931,20 +1035,21 @@ export async function startWorker(
|
|
|
931
1035
|
|
|
932
1036
|
// DES-CRON-VIA-BULLMQ-SCHEDULER live edit (OQ-008): watch the triggers file and re-reconcile schedulers
|
|
933
1037
|
// on change, so an operator's add/edit/delete of a cron trigger takes effect without a worker restart.
|
|
934
|
-
// Only when a triggers file is configured; best-effort
|
|
1038
|
+
// Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
|
|
1039
|
+
// closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
|
|
935
1040
|
if (config.triggersFile) {
|
|
936
|
-
watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared);
|
|
1041
|
+
extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared));
|
|
937
1042
|
}
|
|
938
1043
|
|
|
939
1044
|
// REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
|
|
940
1045
|
// an operator's add/delete of a pause window takes effect without a worker restart. A bad edit is kept out.
|
|
941
1046
|
if (config.pauseWindowsFile) {
|
|
942
|
-
watchPauseWindowsFile(config, pauseWindows, log);
|
|
1047
|
+
extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log));
|
|
943
1048
|
}
|
|
944
1049
|
|
|
945
1050
|
// Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
|
|
946
1051
|
if (config.scopedLimitsFile) {
|
|
947
|
-
watchScopedLimitsFile(config, scopedLimits, log);
|
|
1052
|
+
extraClosers.push(watchScopedLimitsFile(config, scopedLimits, log));
|
|
948
1053
|
}
|
|
949
1054
|
|
|
950
1055
|
log("worker_started", {
|