@edgehero/pi-dispatch 2.0.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +44 -7
- package/README.md +14 -6
- package/deploy/com.pi-dispatch.worker.plist +1 -1
- package/deploy/docker-compose.yml +12 -0
- package/deploy/egress-proxy.conf +28 -3
- package/deploy/pi-dispatch-egress-proxy.container +8 -2
- package/deploy/worker-env-wrapper.cmd +1 -1
- package/deploy/worker-env-wrapper.sh +3 -3
- package/package.json +9 -2
- package/src/allocation.mjs +731 -0
- package/src/backends.mjs +243 -0
- package/src/budget.mjs +40 -4
- package/src/cli.mjs +222 -11
- package/src/config.mjs +126 -5
- package/src/daemon-facts.mjs +3 -0
- package/src/deployment-venue.mjs +1 -0
- package/src/doctor.mjs +2316 -183
- package/src/dollar-budget.mjs +373 -0
- package/src/dollar-fingerprint.mjs +83 -0
- package/src/egress-cli.mjs +316 -0
- package/src/egress-proxy-state.mjs +35 -5
- package/src/egress.mjs +16 -3
- package/src/env-allowlist.mjs +142 -18
- package/src/env-file.mjs +194 -25
- package/src/envelope.mjs +413 -0
- package/src/exit-code.mjs +22 -0
- package/src/fleet-lease.mjs +85 -25
- package/src/get-token.mjs +16 -5
- package/src/git-dirty.mjs +67 -0
- package/src/github-app-setup.mjs +6 -3
- package/src/github-host.mjs +5 -3
- package/src/host-pi.mjs +19 -3
- package/src/identity.mjs +2 -1
- package/src/image-preflight.mjs +98 -24
- package/src/image-ref.mjs +37 -0
- package/src/import-pi.mjs +4 -2
- package/src/index.mjs +407 -62
- package/src/init.mjs +18 -0
- package/src/job-id.mjs +26 -3
- package/src/live-probes.mjs +24 -9
- package/src/model-catalog.mjs +297 -0
- package/src/model-endpoints.mjs +649 -0
- package/src/model-ref.mjs +151 -0
- package/src/models-json.mjs +262 -0
- package/src/money.mjs +144 -0
- package/src/octokit-log.mjs +65 -0
- package/src/outbox-plan.mjs +218 -0
- package/src/outbox.mjs +29 -9
- package/src/output-cap.mjs +157 -0
- package/src/packages.mjs +2 -2
- package/src/pause-windows.mjs +81 -2
- package/src/pi-model-loader.mjs +77 -0
- package/src/podman-stack.mjs +16 -3
- package/src/portfolio-snapshot.mjs +304 -0
- package/src/prepare-local.mjs +247 -12
- package/src/prepare.mjs +35 -3
- package/src/pricing.mjs +9 -5
- package/src/priorities.mjs +569 -0
- package/src/processor.mjs +603 -173
- package/src/project-id.mjs +17 -0
- package/src/projects.mjs +238 -0
- package/src/provider-key.mjs +32 -7
- package/src/provider-steering.mjs +214 -59
- package/src/queue.mjs +111 -6
- package/src/reserved-env.mjs +30 -0
- package/src/run-container.mjs +59 -5
- package/src/run-history.mjs +379 -24
- package/src/run-mirror.mjs +30 -0
- package/src/runtime-settings.mjs +104 -9
- package/src/schedules.mjs +33 -1
- package/src/scoped-limits.mjs +447 -27
- package/src/secrets.mjs +2 -1
- package/src/service.mjs +15 -4
- package/src/session-store.mjs +131 -6
- package/src/start.mjs +528 -40
- package/src/subscriptions.mjs +7 -3
- package/src/triggers-file.mjs +65 -4
- package/src/triggers.mjs +140 -9
- package/src/up.mjs +308 -34
- package/src/valkey-endpoint.mjs +3 -2
package/src/start.mjs
CHANGED
|
@@ -11,6 +11,7 @@ import { makeGitHubAuth } from "./get-token.mjs";
|
|
|
11
11
|
import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING } from "./processor.mjs";
|
|
12
12
|
import { transientError } from "./transient.mjs";
|
|
13
13
|
import { makeGitHubHost } from "./github-host.mjs";
|
|
14
|
+
import { githubFailureFields } from "./octokit-log.mjs";
|
|
14
15
|
import { makeGitLabAuth } from "./gitlab-auth.mjs";
|
|
15
16
|
import { makeGitLabHost } from "./gitlab-host.mjs";
|
|
16
17
|
import { makeForgejoAuth } from "./forgejo-auth.mjs";
|
|
@@ -18,14 +19,18 @@ import { makeForgejoHost } from "./forgejo-host.mjs";
|
|
|
18
19
|
import { makeAzureAuth } from "./azure-auth.mjs";
|
|
19
20
|
import { makeAzureHost } from "./azure-host.mjs";
|
|
20
21
|
import { makeEgressPreflight } from "./egress.mjs";
|
|
21
|
-
import { checkSlotKey, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
|
|
22
|
+
import { checkSlotKey, endpointSlotKey, hash16, makeClaimSweeper, makeFleetLease, makeScopeClaimSweeper, scopeSlotKey } from "./fleet-lease.mjs";
|
|
23
|
+
import { MAX_SLOTS, loadModelEndpoints, modelEndpointsPath, readOverlayModels } from "./model-endpoints.mjs";
|
|
24
|
+
import { builtinModel, checkModelsKnown } from "./model-catalog.mjs";
|
|
22
25
|
import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
|
|
23
26
|
import { cronFingerprint } from "./fingerprint.mjs";
|
|
24
27
|
import { makeHostRegistry } from "./host-registry.mjs";
|
|
25
28
|
import { makeImagePreflight } from "./image-preflight.mjs";
|
|
26
|
-
import { createWorker, JOB_TIMEOUT_MS } from "./index.mjs";
|
|
29
|
+
import { createWorker, JOB_TIMEOUT_MS, STALLED_FAILED_REASON } from "./index.mjs";
|
|
27
30
|
import { BOOT_REFUSING_JOB_USER_CAUSES, DAEMON_FACTS_TIMEOUT_MS, jobUserRefusal, makeDaemonFactsReader, makeJobUserResolver, relabelsPrivateMounts, resolveImageUser } from "./job-user.mjs";
|
|
28
31
|
import { makeCollectChain } from "./outbox.mjs";
|
|
32
|
+
import { makeCollectPlan } from "./outbox-plan.mjs";
|
|
33
|
+
import { makePortfolioSnapshot } from "./portfolio-snapshot.mjs";
|
|
29
34
|
import { containerPackagePaths, readStageManifest } from "./packages.mjs";
|
|
30
35
|
import { makeCleanup, makeForgePreparers, makePrepareWorkspace } from "./prepare.mjs";
|
|
31
36
|
import { listRunningSandboxes, makeSandboxNetworkSweeper, makeSandboxRuntimeWatch } from "./sandbox.mjs";
|
|
@@ -33,10 +38,14 @@ import { makeRetentionSweep } from "./retention-sweep.mjs";
|
|
|
33
38
|
import { makeSandboxReaper } from "./sandbox-store.mjs";
|
|
34
39
|
import { makeSessionStore } from "./session-store.mjs";
|
|
35
40
|
import { scrubCredentials } from "./redact.mjs";
|
|
36
|
-
import { makeCheckOnceSpent, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
|
|
41
|
+
import { makeCheckOnceSpent, makeCheckPortfolioFlag, makeCheckWaitSkew, makeDisarmOnce } from "./triggers-file.mjs";
|
|
37
42
|
import { WATCH_DEBOUNCE_MS, changedWhileArming, makeWatchCloser, readBeforeArming } from "./watch-closer.mjs";
|
|
38
43
|
import { loadPauseWindows, pauseUntilMs } from "./pause-windows.mjs";
|
|
39
|
-
import { loadScopedLimits,
|
|
44
|
+
import { checkProjectRows, danglingProjectRows, dollarRowsWithoutCap, loadScopedLimits, scopeClaimRows } from "./scoped-limits.mjs";
|
|
45
|
+
import { escapeControls, loadProjects, projectOf, projectsFingerprint } from "./projects.mjs";
|
|
46
|
+
import { envelopeDigest, envelopeInsideJobPaths, loadEnvelopeChecked } from "./envelope.mjs";
|
|
47
|
+
import { NO_ENVELOPE_FINGERPRINT, makeAllocationAudit, makeAllocationLogReaper, makeAllocationState } from "./allocation.mjs";
|
|
48
|
+
import { optionalUsdMicros } from "./money.mjs";
|
|
40
49
|
import { makeOnFailure } from "./on-failure.mjs";
|
|
41
50
|
import { makeWaitChecker } from "./wait-check.mjs";
|
|
42
51
|
import { makeWaitState } from "./wait-state.mjs";
|
|
@@ -44,17 +53,18 @@ import { hostQueueName, makeQueue } from "./queue.mjs";
|
|
|
44
53
|
import { endpointShown, makeDockerEndpointResolver, makeLocalBackend, makeReaper, makeStopContainer, quotedShown } from "./backend-local.mjs";
|
|
45
54
|
import { NETNS_KEEPER_MIN_AGE_MS, NETNS_KEEPER_YOUNG_MARGIN_MS, runtimeFromFacts } from "./netns-keeper.mjs";
|
|
46
55
|
import { makeBackendRegistry, reapAll, resolveBackendName } from "./backend-registry.mjs";
|
|
47
|
-
import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
|
|
56
|
+
import { DEFAULT_BACKEND, DOCKER_ENDPOINT_LOCAL, PODMAN_ADDS_NO_MOUNTS, PODMAN_BACKEND, PODMAN_BOUNDS_DELEGATED, PODMAN_SERVICE_LOCAL, backendFor, isPerMachineHost, observationRefusalIsTransient, observationRefusals, unobservedFloor } from "./backends.mjs";
|
|
48
57
|
import { PODMAN_BOOT_REFUSING_CAUSES, PODMAN_INFO_TIMEOUT_MS, cachedPodmanInfo, decidePodmanJobUser, makePodmanBackend, makePodmanInfoReader, makePodmanReaper, observePodman, podmanConfRefusal, unavailableFor, podmanConfWidening, podmanJobUserRefusal, resolvePodmanImageUser } from "./backend-podman.mjs";
|
|
49
58
|
import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootfulMemory, observeHost, observeRootfulConf, readRootfulService, rootfulConfRefusal, rootfulConfRetries, rootfulUnreadList, runtimeObservationKey } from "./runtime-observations.mjs";
|
|
50
59
|
|
|
51
60
|
import { makeRunContainer } from "./run-container.mjs";
|
|
52
61
|
import { resolveProviderCredential } from "./env-allowlist.mjs";
|
|
53
62
|
import { makeSecretsResolver } from "./secrets.mjs";
|
|
54
|
-
import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeRecordWriter, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
|
|
55
|
-
import { makeRunMirror } from "./run-mirror.mjs";
|
|
56
|
-
import {
|
|
57
|
-
import {
|
|
63
|
+
import { buildRecord, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
|
|
64
|
+
import { makeRunMirror, readMirroredRecord } from "./run-mirror.mjs";
|
|
65
|
+
import { readOverlay, resolveSettings } from "./runtime-settings.mjs";
|
|
66
|
+
import { usdFingerprint } from "./dollar-fingerprint.mjs";
|
|
67
|
+
import { authoredCron, envelopeJobPaths, loadSchedules, servedSchedules } from "./schedules.mjs";
|
|
58
68
|
import { makeStallGuard } from "./scheduler-stall-guard.mjs";
|
|
59
69
|
|
|
60
70
|
|
|
@@ -164,7 +174,7 @@ export async function settleWithin(promise, ms, fallback) {
|
|
|
164
174
|
* schedulers. The FSWatcher is unref'd (the debounce it arms is NOT), and the returned closer is what
|
|
165
175
|
* `startWorker` registers so the watch dies with the worker that armed it (issue #295).
|
|
166
176
|
*/
|
|
167
|
-
function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot) {
|
|
177
|
+
function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot, afterReload = () => {}) {
|
|
168
178
|
const path = config.triggersFile;
|
|
169
179
|
const dir = dirname(path) || ".";
|
|
170
180
|
const file = basename(path);
|
|
@@ -180,7 +190,9 @@ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot)
|
|
|
180
190
|
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
181
191
|
if (changed && changed !== file) return; // only our file (a null name -> reload to be safe)
|
|
182
192
|
clearTimeout(handles.timer);
|
|
183
|
-
|
|
193
|
+
// Issue #504 part B: a cron folder or a skills dir an edit adds is a new job path, so the envelope's place is
|
|
194
|
+
// judged again after every triggers reload.
|
|
195
|
+
handles.timer = setTimeout(() => void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }).then(() => afterReload(closer.reloadLog)), WATCH_DEBOUNCE_MS);
|
|
184
196
|
});
|
|
185
197
|
handles.watcher.unref?.();
|
|
186
198
|
log("triggers_watching", { path });
|
|
@@ -192,7 +204,7 @@ function watchTriggersFile(config, queue, log, ref, registry, tz, fleet, atBoot)
|
|
|
192
204
|
// changed.
|
|
193
205
|
if (changedWhileArming(handles, readFile)) {
|
|
194
206
|
log("triggers_reread_after_arming", { path });
|
|
195
|
-
void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet });
|
|
207
|
+
void reloadSchedules(config, queue, { log: closer.reloadLog, ref, registry, tz, fleet }).then(() => afterReload(closer.reloadLog));
|
|
196
208
|
}
|
|
197
209
|
return closer;
|
|
198
210
|
}
|
|
@@ -246,13 +258,74 @@ function watchPauseWindowsFile(config, ref, log, atBoot) {
|
|
|
246
258
|
* last-good property carries its own test). A bad edit keeps `ref.current` untouched and logs
|
|
247
259
|
* `scoped_limits_reload_invalid` -- the pause-windows posture, INT-SCOPED-LIMITS-FILE-CONTRACT.
|
|
248
260
|
*/
|
|
249
|
-
export function reloadScopedLimits(config, ref, log) {
|
|
261
|
+
export function reloadScopedLimits(config, ref, log, deploymentCap = null, pair = null) {
|
|
250
262
|
try {
|
|
251
|
-
|
|
263
|
+
const next = loadScopedLimits(config);
|
|
264
|
+
// Issue #499 part B: the new limits are checked against projects.json (`pair`, null only on a bare test wiring);
|
|
265
|
+
// a row naming a missing project keeps the last good limits, the boot rule held live.
|
|
266
|
+
const other = pair ? pairWith(config, next, pair, "limits") : null;
|
|
267
|
+
ref.current = next;
|
|
252
268
|
log("scoped_limits_reloaded", { count: ref.current.length });
|
|
269
|
+
if (deploymentCap) warnDollarRowsWithoutCap(ref.current, deploymentCap(), log);
|
|
270
|
+
if (other) {
|
|
271
|
+
pair.projects.current = other;
|
|
272
|
+
log("projects_reloaded", { count: other.length, with: "scoped-limits" });
|
|
273
|
+
}
|
|
253
274
|
} catch (err) {
|
|
254
275
|
log("scoped_limits_reload_invalid", { reason: err?.message });
|
|
276
|
+
return;
|
|
277
|
+
}
|
|
278
|
+
// Issue #504 part B: the envelope's floors are judged against both files, so a committed edit re-judges it.
|
|
279
|
+
pair?.afterCommit?.();
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/**
|
|
283
|
+
* The two files a project row joins (issue #499 part B): scoped-limits.json names `project:<id>`, projects.json defines
|
|
284
|
+
* the id. `{ limits, projects }` are the two live refs, and `deploymentCap` the merged per-job cap thunk the
|
|
285
|
+
* dollar-rows-without-cap warning reads (null on a bare wiring). A reload of either file is judged as a PAIR (`pairWith`), so the
|
|
286
|
+
* two live lists never disagree and a correct pair applies whatever order its files were saved in.
|
|
287
|
+
*/
|
|
288
|
+
export function makeProjectPair(limits, projects, deploymentCap = null) {
|
|
289
|
+
return { limits, projects, deploymentCap };
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
/**
|
|
293
|
+
* Judge one side's new list (`next`, already loaded) against the other side, statelessly (issue #499 part B):
|
|
294
|
+
* 1. against the other file AS IT IS ON DISK: when that loads and the two agree, both are taken together, and the
|
|
295
|
+
* other side's list is returned for the caller to commit too (only when it differs from the live one). This is
|
|
296
|
+
* what makes a rename (`shop` to `store` in both files) or a project added with its row apply in EITHER save
|
|
297
|
+
* order: the second save sees the first file already on disk.
|
|
298
|
+
* 2. else against the other side's LIVE list: when they agree, this side alone is taken (null returned). This is the
|
|
299
|
+
* path when the other file is mid-edit and does not load.
|
|
300
|
+
* 3. else a `configError` naming the row, its index and both files, so the caller keeps its last good list.
|
|
301
|
+
*/
|
|
302
|
+
function pairWith(config, next, pair, side) {
|
|
303
|
+
const limitsSide = side === "limits";
|
|
304
|
+
let disk = null;
|
|
305
|
+
try {
|
|
306
|
+
disk = limitsSide ? loadProjects(config) : loadScopedLimits(config);
|
|
307
|
+
} catch {
|
|
308
|
+
// The other file does not load right now: its own watcher says so. Judge against its live list alone.
|
|
255
309
|
}
|
|
310
|
+
const agree = (limits, projects) => danglingProjectRows(limits, projects).length === 0;
|
|
311
|
+
if (disk !== null && (limitsSide ? agree(next, disk) : agree(disk, next))) {
|
|
312
|
+
const live = limitsSide ? pair.projects.current : pair.limits.current;
|
|
313
|
+
return JSON.stringify(disk) === JSON.stringify(live) ? null : disk;
|
|
314
|
+
}
|
|
315
|
+
const liveOther = limitsSide ? pair.projects.current : pair.limits.current;
|
|
316
|
+
if (limitsSide) checkProjectRows(next, liveOther, config.scopedLimitsFile, config.projectsFile ?? null);
|
|
317
|
+
else checkProjectRows(liveOther, next, config.scopedLimitsFile, config.projectsFile ?? null);
|
|
318
|
+
return null;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/**
|
|
322
|
+
* PR #549's review: a `scoped-limits.json` dollar row on a deployment with no per-job cap (env and overlay merged)
|
|
323
|
+
* refuses every job it applies to as `config-refused`, unless that job's trigger sets `run.maxCostUsd`. So it is a
|
|
324
|
+
* WARNING at load and at each reload, never a refusal: the rows by index and kind, never a scope string.
|
|
325
|
+
*/
|
|
326
|
+
export function warnDollarRowsWithoutCap(limits, deploymentMaxCostUsd, log) {
|
|
327
|
+
const rows = dollarRowsWithoutCap(limits, deploymentMaxCostUsd);
|
|
328
|
+
if (rows.length > 0) log("scoped_limits_dollar_rows_without_cap", { rows });
|
|
256
329
|
}
|
|
257
330
|
|
|
258
331
|
/**
|
|
@@ -260,7 +333,7 @@ export function reloadScopedLimits(config, ref, log) {
|
|
|
260
333
|
* for atomic tmp+rename robustness, filtered to the one basename, debounced. Best-effort; the FSWatcher is
|
|
261
334
|
* unref'd and the returned closer stops the watch with the worker (issue #295).
|
|
262
335
|
*/
|
|
263
|
-
function watchScopedLimitsFile(config, ref, log, atBoot) {
|
|
336
|
+
function watchScopedLimitsFile(config, ref, log, atBoot, deploymentCap = null, pair = null) {
|
|
264
337
|
const path = config.scopedLimitsFile;
|
|
265
338
|
const dir = dirname(path) || ".";
|
|
266
339
|
const file = basename(path);
|
|
@@ -273,7 +346,7 @@ function watchScopedLimitsFile(config, ref, log, atBoot) {
|
|
|
273
346
|
if (handles.closed) return; // see makeWatchCloser: by construction, not by a delivery rule
|
|
274
347
|
if (changed && changed !== file) return;
|
|
275
348
|
clearTimeout(handles.timer);
|
|
276
|
-
handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
|
|
349
|
+
handles.timer = setTimeout(() => reloadScopedLimits(config, ref, closer.reloadLog, deploymentCap, pair), WATCH_DEBOUNCE_MS);
|
|
277
350
|
});
|
|
278
351
|
handles.watcher.unref?.();
|
|
279
352
|
log("scoped_limits_watching", { path });
|
|
@@ -282,7 +355,188 @@ function watchScopedLimitsFile(config, ref, log, atBoot) {
|
|
|
282
355
|
}
|
|
283
356
|
if (changedWhileArming(handles, readFile)) {
|
|
284
357
|
log("scoped_limits_reread_after_arming", { path });
|
|
285
|
-
reloadScopedLimits(config, ref, closer.reloadLog);
|
|
358
|
+
reloadScopedLimits(config, ref, closer.reloadLog, deploymentCap, pair);
|
|
359
|
+
}
|
|
360
|
+
return closer;
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/**
|
|
364
|
+
* The projects reload (issue #499), exported apart from its watcher for `reloadScopedLimits`' reason: last-good is
|
|
365
|
+
* testable without fs.watch. A bad edit keeps `ref.current` and logs `projects_reload_invalid`; a good one swaps it,
|
|
366
|
+
* so the next pickup resolves against the new membership. A job already past its pickup keeps the id it was given.
|
|
367
|
+
* The reason is the loader's message, which never quotes a project's `name` (projects.mjs).
|
|
368
|
+
*/
|
|
369
|
+
export function reloadProjects(config, ref, log, pair = null) {
|
|
370
|
+
try {
|
|
371
|
+
const next = loadProjects(config);
|
|
372
|
+
// Issue #499 part B: an edit that drops (or renames) a project a scoped-limits row still names is kept out, and the
|
|
373
|
+
// last good projects stay, unless scoped-limits.json on disk already agrees with it (`pairWith`).
|
|
374
|
+
const other = pair ? pairWith(config, next, pair, "projects") : null;
|
|
375
|
+
ref.current = next;
|
|
376
|
+
log("projects_reloaded", { count: ref.current.length });
|
|
377
|
+
if (other) {
|
|
378
|
+
pair.limits.current = other;
|
|
379
|
+
log("scoped_limits_reloaded", { count: other.length, with: "projects" });
|
|
380
|
+
// Every limits list that goes live is warned on, whichever file's reload committed it.
|
|
381
|
+
if (pair.deploymentCap) warnDollarRowsWithoutCap(other, pair.deploymentCap(), log);
|
|
382
|
+
}
|
|
383
|
+
} catch (err) {
|
|
384
|
+
// Escaped (PR #569's review): the parser's own refusals already are, and an fs error quoting the path is too.
|
|
385
|
+
log("projects_reload_invalid", { reason: escapeControls(err?.message) });
|
|
386
|
+
return;
|
|
387
|
+
}
|
|
388
|
+
// Issue #504 part B: as `reloadScopedLimits` does, so the envelope follows a projects edit in either save order.
|
|
389
|
+
pair?.afterCommit?.();
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/**
|
|
393
|
+
* Watch the projects file (issue #499) as the scoped-limits watcher does: the directory, filtered to the one basename,
|
|
394
|
+
* debounced, the boot read as the arming baseline. The closer joins `extraClosers`, so the watch stops with the worker
|
|
395
|
+
* (DES-WATCHERS-CLOSE-WITH-THE-WORKER).
|
|
396
|
+
*/
|
|
397
|
+
function watchProjectsFile(config, ref, log, atBoot, pair = null) {
|
|
398
|
+
const path = config.projectsFile;
|
|
399
|
+
const dir = dirname(path) || ".";
|
|
400
|
+
const file = basename(path);
|
|
401
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
402
|
+
const closer = makeWatchCloser(handles, log);
|
|
403
|
+
const readFile = () => readFileSync(path, "utf8");
|
|
404
|
+
readBeforeArming(handles, readFile, atBoot);
|
|
405
|
+
try {
|
|
406
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
407
|
+
if (handles.closed) return;
|
|
408
|
+
if (changed && changed !== file) return;
|
|
409
|
+
clearTimeout(handles.timer);
|
|
410
|
+
handles.timer = setTimeout(() => reloadProjects(config, ref, closer.reloadLog, pair), WATCH_DEBOUNCE_MS);
|
|
411
|
+
});
|
|
412
|
+
handles.watcher.unref?.();
|
|
413
|
+
log("projects_watching", { path });
|
|
414
|
+
} catch (err) {
|
|
415
|
+
log("projects_watch_unavailable", { reason: err?.message });
|
|
416
|
+
}
|
|
417
|
+
if (changedWhileArming(handles, readFile)) {
|
|
418
|
+
log("projects_reread_after_arming", { path });
|
|
419
|
+
reloadProjects(config, ref, closer.reloadLog, pair);
|
|
420
|
+
}
|
|
421
|
+
return closer;
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
/**
|
|
425
|
+
* `envelopeJobPaths` as a thunk that keeps its last good answer (issue #504 part B): a triggers file caught mid-edit
|
|
426
|
+
* must not turn a containment check into a failure of its own, and the last parse that succeeded is what the running
|
|
427
|
+
* schedulers were built from. The first call has no last good answer and throws, which at boot refuses.
|
|
428
|
+
*/
|
|
429
|
+
export function makeEnvelopeJobPaths(config, io = {}) {
|
|
430
|
+
let lastGood = null;
|
|
431
|
+
return () => {
|
|
432
|
+
try {
|
|
433
|
+
lastGood = envelopeJobPaths(config, io);
|
|
434
|
+
} catch (err) {
|
|
435
|
+
if (lastGood === null) throw err;
|
|
436
|
+
}
|
|
437
|
+
return lastGood;
|
|
438
|
+
};
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
/**
|
|
442
|
+
* The envelope reload (issue #504 part B), exported apart from its watcher for `reloadScopedLimits`' reason. `ref` is
|
|
443
|
+
* `{ current, digest }`. The file is judged against the LIVE projects and scoped limits (its floors name projects and
|
|
444
|
+
* sit under project rows) and the job paths of the moment, so a reload is paired with both files: their own reloads
|
|
445
|
+
* call this again once they commit (`pair.afterCommit`), and an envelope edit that needs a projects edit applies in
|
|
446
|
+
* either save order. A bad edit, or one that puts the file inside a job path, keeps the last good envelope and logs
|
|
447
|
+
* `envelope_reload_invalid`; the last good copy can never be one a job wrote, because every reload re-runs the
|
|
448
|
+
* containment check. A changed digest logs `envelope_reloaded` and calls `onChange` (the re-base, or the mismatch).
|
|
449
|
+
*/
|
|
450
|
+
export function reloadEnvelope(config, ref, log, { projects, limits, maxCostMicros = () => null, jobPaths, onChange = () => {} }) {
|
|
451
|
+
let next;
|
|
452
|
+
try {
|
|
453
|
+
next = loadEnvelopeChecked(config, { projects: projects?.current ?? [], limits: limits?.current ?? [], maxCostMicros: maxCostMicros(), jobPaths: jobPaths() });
|
|
454
|
+
} catch (err) {
|
|
455
|
+
log("envelope_reload_invalid", { reason: escapeControls(err?.message) });
|
|
456
|
+
return;
|
|
457
|
+
}
|
|
458
|
+
const digest = next === null ? null : envelopeDigest(next);
|
|
459
|
+
const changed = digest !== ref.digest;
|
|
460
|
+
ref.current = next;
|
|
461
|
+
ref.digest = digest;
|
|
462
|
+
if (!changed) return;
|
|
463
|
+
log("envelope_reloaded", { digest });
|
|
464
|
+
onChange();
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
/**
|
|
468
|
+
* Watch the envelope file (issue #504 part B) as the projects watcher does: the directory, filtered to the one
|
|
469
|
+
* basename, debounced, the boot read as the arming baseline, the closer joining `extraClosers`.
|
|
470
|
+
*/
|
|
471
|
+
function watchEnvelopeFile(config, ref, log, atBoot, ctx) {
|
|
472
|
+
const path = config.envelopeFile;
|
|
473
|
+
const dir = dirname(path) || ".";
|
|
474
|
+
const file = basename(path);
|
|
475
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
476
|
+
const closer = makeWatchCloser(handles, log);
|
|
477
|
+
const readFile = () => readFileSync(path, "utf8");
|
|
478
|
+
readBeforeArming(handles, readFile, atBoot);
|
|
479
|
+
try {
|
|
480
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
481
|
+
if (handles.closed) return;
|
|
482
|
+
if (changed && changed !== file) return;
|
|
483
|
+
clearTimeout(handles.timer);
|
|
484
|
+
handles.timer = setTimeout(() => reloadEnvelope(config, ref, closer.reloadLog, ctx), WATCH_DEBOUNCE_MS);
|
|
485
|
+
});
|
|
486
|
+
handles.watcher.unref?.();
|
|
487
|
+
log("envelope_watching", { path });
|
|
488
|
+
} catch (err) {
|
|
489
|
+
log("envelope_watch_unavailable", { reason: err?.message });
|
|
490
|
+
}
|
|
491
|
+
if (changedWhileArming(handles, readFile)) {
|
|
492
|
+
log("envelope_reread_after_arming", { path });
|
|
493
|
+
reloadEnvelope(config, ref, closer.reloadLog, ctx);
|
|
494
|
+
}
|
|
495
|
+
return closer;
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/**
|
|
499
|
+
* The model-endpoints reload (issue #503), exported apart from its watcher for `reloadScopedLimits`' reason: the
|
|
500
|
+
* last-good property is testable without fs.watch. A bad edit keeps `ref.current` and logs
|
|
501
|
+
* `model_endpoints_reload_invalid`; a good one swaps it, so a `slots` edit applies to the next pickup.
|
|
502
|
+
*/
|
|
503
|
+
export function reloadModelEndpoints(config, ref, log) {
|
|
504
|
+
try {
|
|
505
|
+
ref.current = loadModelEndpoints(config);
|
|
506
|
+
log("model_endpoints_reloaded", { count: ref.current.length });
|
|
507
|
+
} catch (err) {
|
|
508
|
+
log("model_endpoints_reload_invalid", { reason: err?.message });
|
|
509
|
+
}
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
/**
|
|
513
|
+
* Watch the model-endpoints file (issue #503) as the scoped-limits watcher does. Armed only when the file is named
|
|
514
|
+
* or exists at boot: the default path is the deployment folder, and a worker started elsewhere must not watch an
|
|
515
|
+
* arbitrary directory for a file nobody declared. A default file created later is read at the next restart.
|
|
516
|
+
*/
|
|
517
|
+
function watchModelEndpointsFile(config, ref, log, atBoot) {
|
|
518
|
+
const { path } = modelEndpointsPath(config);
|
|
519
|
+
const dir = dirname(path) || ".";
|
|
520
|
+
const file = basename(path);
|
|
521
|
+
const handles = { watcher: null, timer: null, closed: false };
|
|
522
|
+
const closer = makeWatchCloser(handles, log);
|
|
523
|
+
const readFile = () => readFileSync(path, "utf8");
|
|
524
|
+
readBeforeArming(handles, readFile, atBoot);
|
|
525
|
+
try {
|
|
526
|
+
handles.watcher = watch(dir, (_event, changed) => {
|
|
527
|
+
if (handles.closed) return;
|
|
528
|
+
if (changed && changed !== file) return;
|
|
529
|
+
clearTimeout(handles.timer);
|
|
530
|
+
handles.timer = setTimeout(() => reloadModelEndpoints(config, ref, closer.reloadLog), WATCH_DEBOUNCE_MS);
|
|
531
|
+
});
|
|
532
|
+
handles.watcher.unref?.();
|
|
533
|
+
log("model_endpoints_watching", { path });
|
|
534
|
+
} catch (err) {
|
|
535
|
+
log("model_endpoints_watch_unavailable", { reason: err?.message });
|
|
536
|
+
}
|
|
537
|
+
if (changedWhileArming(handles, readFile)) {
|
|
538
|
+
log("model_endpoints_reread_after_arming", { path });
|
|
539
|
+
reloadModelEndpoints(config, ref, closer.reloadLog);
|
|
286
540
|
}
|
|
287
541
|
return closer;
|
|
288
542
|
}
|
|
@@ -330,6 +584,8 @@ export async function startWorker(
|
|
|
330
584
|
makeLogSink: makeLogSinkFn = makeLogSink,
|
|
331
585
|
makeRecordWriter: makeRecordWriterFn = makeRecordWriter,
|
|
332
586
|
makeRunMirror: makeRunMirrorFn = makeRunMirror,
|
|
587
|
+
// The fleet copy's by-id read, seamed so a wiring test can answer it without a live mirror.
|
|
588
|
+
readMirroredRecord: readMirroredRecordFn = readMirroredRecord,
|
|
333
589
|
makeLogReaper: makeLogReaperFn = makeLogReaper,
|
|
334
590
|
makeSandboxReaper: makeSandboxReaperFn = makeSandboxReaper,
|
|
335
591
|
makeSandboxNetworkSweeper: makeSandboxNetworkSweeperFn = makeSandboxNetworkSweeper,
|
|
@@ -341,6 +597,7 @@ export async function startWorker(
|
|
|
341
597
|
makeSecretsResolver: makeSecretsResolverFn = makeSecretsResolver,
|
|
342
598
|
makeImagePreflight: makeImagePreflightFn = makeImagePreflight,
|
|
343
599
|
makeScopeClaimSweeper: makeScopeClaimSweeperFn = makeScopeClaimSweeper,
|
|
600
|
+
makeClaimSweeper: makeClaimSweeperFn = makeClaimSweeper,
|
|
344
601
|
makeHostRegistry: makeHostRegistryFn = makeHostRegistry,
|
|
345
602
|
makeEgressPreflight: makeEgressPreflightFn = makeEgressPreflight,
|
|
346
603
|
// Which docker endpoint this host's CLI resolves (issue #278). A seam because the real one spawns the
|
|
@@ -391,6 +648,13 @@ export async function startWorker(
|
|
|
391
648
|
// literal address every Valkey client of this worker then connects to. A seam: the real one reads /proc, probes
|
|
392
649
|
// this host's addresses and resolves the name, none of which belongs in a wiring test.
|
|
393
650
|
judgeValkey = defaultJudgeValkey,
|
|
651
|
+
// The scoped-limits watcher, injectable so a test can see what it is armed with (PR #549's review: the cap
|
|
652
|
+
// thunk the reload warning reads). Production passes nothing and gets the real watcher.
|
|
653
|
+
watchScopedLimits: watchScopedLimitsFn = watchScopedLimitsFile,
|
|
654
|
+
// The projects watcher (issue #499), injectable for the same reason: a test sees it armed and closed.
|
|
655
|
+
watchProjects: watchProjectsFn = watchProjectsFile,
|
|
656
|
+
// The envelope watcher (issue #504 part B), injectable for the same reason.
|
|
657
|
+
watchEnvelope: watchEnvelopeFn = watchEnvelopeFile,
|
|
394
658
|
} = {},
|
|
395
659
|
) {
|
|
396
660
|
const config = loadConfigFn(env);
|
|
@@ -463,7 +727,7 @@ export async function startWorker(
|
|
|
463
727
|
// this race lives in is between these lines and the arming a thousand lines below -- the endpoint probe,
|
|
464
728
|
// forge auth, the reaper, Valkey -- not the microseconds around the arming itself, which is what a first
|
|
465
729
|
// attempt measured. `null` where a file is not configured, which reads as "nothing to compare".
|
|
466
|
-
const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null };
|
|
730
|
+
const atBoot = { triggers: null, pauseWindows: null, scopedLimits: null, projects: null, modelEndpoints: null, envelope: null };
|
|
467
731
|
const recording = (into, path) => ({
|
|
468
732
|
readFileSync: (file, enc) => {
|
|
469
733
|
const text = readFileSync(file, enc);
|
|
@@ -482,6 +746,55 @@ export async function startWorker(
|
|
|
482
746
|
// ref for the live-reload watcher, [] when unset (the folder mutex is code and needs no file).
|
|
483
747
|
const scopedLimits = { current: loadScopedLimits(config, recording("scopedLimits", config.scopedLimitsFile)) };
|
|
484
748
|
|
|
749
|
+
// Issue #499: the projects file, same posture (INT-PROJECTS-FILE-CONTRACT): a bad file refuses boot with the operator
|
|
750
|
+
// present, a mutable ref for the live reload, [] when unset. The pickup gate reads it once per pickup, beside the
|
|
751
|
+
// limits snapshot, and the id it resolves there is the one the job's record carries.
|
|
752
|
+
const projects = { current: loadProjects(config, recording("projects", config.projectsFile)) };
|
|
753
|
+
// Issue #499 part B: a `project:<id>` row whose id is not a project refuses BOOT, naming the row and the id: it would
|
|
754
|
+
// read as a cap on a group that no job can belong to. The live reloads of either file hold the same rule (`pair`).
|
|
755
|
+
checkProjectRows(scopedLimits.current, projects.current, config.scopedLimitsFile, config.projectsFile ?? null);
|
|
756
|
+
|
|
757
|
+
// Issue #504 part B: the allocation envelope (INT-ENVELOPE-FILE-CONTRACT), same posture: a bad file refuses boot with
|
|
758
|
+
// the operator present, before any Valkey contact. It is judged against the projects and scoped limits just loaded
|
|
759
|
+
// (its floors name projects and sit under project rows), needs the merged per-job cost cap (each governed job
|
|
760
|
+
// reserves it against its share), and must lie outside every host path a job container can see, which is checked
|
|
761
|
+
// with this load and again with every reload. A mutable ref `{ current, digest }`; null when PI_ENVELOPE_FILE is
|
|
762
|
+
// unset, and then nothing below governs anything.
|
|
763
|
+
const envelopeCap = () => {
|
|
764
|
+
try {
|
|
765
|
+
const s = resolveSettings(config, readOverlay(config.settingsFile));
|
|
766
|
+
return optionalUsdMicros(s.invalid ? config.maxCostUsd : s.maxCostUsd, "maxCostUsd");
|
|
767
|
+
} catch {
|
|
768
|
+
return null;
|
|
769
|
+
}
|
|
770
|
+
};
|
|
771
|
+
const envelopeJobPathsNow = makeEnvelopeJobPaths(config);
|
|
772
|
+
const envelope = { current: null, digest: null };
|
|
773
|
+
if (config.envelopeFile !== null && config.envelopeFile !== undefined) {
|
|
774
|
+
envelope.current = loadEnvelopeChecked(config, { projects: projects.current, limits: scopedLimits.current, maxCostMicros: envelopeCap(), jobPaths: envelopeJobPathsNow() }, { io: recording("envelope", config.envelopeFile) });
|
|
775
|
+
envelope.digest = envelope.current === null ? null : envelopeDigest(envelope.current);
|
|
776
|
+
log("envelope_loaded", { digest: envelope.digest, window: envelope.current?.window ?? null, delegation: envelope.current?.delegation?.enabled === true });
|
|
777
|
+
}
|
|
778
|
+
// After a triggers reload (a cron folder or a skills dir is a job path): the envelope's place judged again. The live
|
|
779
|
+
// copy is kept either way, since every envelope reload re-runs the check and so never adopts a file a job could have
|
|
780
|
+
// written; this line is what tells the operator that a job can now reach the file.
|
|
781
|
+
const checkEnvelopePlace = (say = log) => {
|
|
782
|
+
if (!envelope.current) return;
|
|
783
|
+
try {
|
|
784
|
+
const inside = envelopeInsideJobPaths(config.envelopeFile, envelopeJobPathsNow());
|
|
785
|
+
if (inside) say("envelope_inside_job_path", { kind: inside.kind });
|
|
786
|
+
} catch (err) {
|
|
787
|
+
say("envelope_inside_job_path", { reason: escapeControls(err?.message) });
|
|
788
|
+
}
|
|
789
|
+
};
|
|
790
|
+
|
|
791
|
+
// Issue #503: the declared model endpoints, same posture (INT-MODEL-ENDPOINTS-FILE-CONTRACT): a bad file refuses
|
|
792
|
+
// boot with the operator present, a mutable ref for the live reload, [] when there is no file. Read per pickup
|
|
793
|
+
// from the ref, so a `slots` edit applies to the next job.
|
|
794
|
+
const modelEndpointsAt = modelEndpointsPath(config);
|
|
795
|
+
const modelEndpoints = { current: loadModelEndpoints(config, recording("modelEndpoints", modelEndpointsAt.path)) };
|
|
796
|
+
const watchModelEndpoints = modelEndpointsAt.explicit || atBoot.modelEndpoints !== null;
|
|
797
|
+
|
|
485
798
|
// Issue #278: WHICH DOCKER DAEMON the job containers' credentials will travel to. Asked of the CLI at boot,
|
|
486
799
|
// AFTER the free file validations above (a wedged CLI costs up to its bound, and must not delay them) and
|
|
487
800
|
// BEFORE forge auth, the reaper, Valkey and the worker -- so a refusal here stops a process that built
|
|
@@ -643,7 +956,9 @@ export async function startWorker(
|
|
|
643
956
|
// `doctor` reports as healthy. The boot posture is unchanged for a DETERMINATE failure, which is what
|
|
644
957
|
// the local-only case is (no `gh` on PATH is `ENOENT`); a TRANSIENT one now leaves a re-resolver
|
|
645
958
|
// behind instead of a permanent null.
|
|
646
|
-
|
|
959
|
+
// Issue #530: the GitHub clients take this worker's logger, so a client's own warning is one more JSON line here and
|
|
960
|
+
// its plain request lines (`GET /user - 401 ...`) are never printed (`octokitLog`).
|
|
961
|
+
const forges = { github: { auth: null, host: makeHost({ log }) } };
|
|
647
962
|
// Per forge kind, what it would take to resolve its auth again: the closure, and the `idOf` its log
|
|
648
963
|
// line needs. Present only while the last attempt failed transiently -- a determinate failure removes
|
|
649
964
|
// it, because retrying a wrong credential is how a deployment pays to be told the same thing twice.
|
|
@@ -670,7 +985,7 @@ export async function startWorker(
|
|
|
670
985
|
// modules throw untagged for a fetch rejection, a transient status and an unparseable body.
|
|
671
986
|
const transient = err?.piDispatchConfig !== true;
|
|
672
987
|
if (transient) authRetries.set(kind, { resolve: () => make(cfg), idOf });
|
|
673
|
-
log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient });
|
|
988
|
+
log(`${kind}_auth_unavailable`, { kind, reason: err?.message, ...githubFailureFields(err), transient });
|
|
674
989
|
}
|
|
675
990
|
};
|
|
676
991
|
|
|
@@ -711,7 +1026,7 @@ export async function startWorker(
|
|
|
711
1026
|
authLastError.set(kind, err);
|
|
712
1027
|
if (err?.piDispatchConfig === true) {
|
|
713
1028
|
authRetries.delete(kind);
|
|
714
|
-
log(`${kind}_auth_unavailable`, { kind, reason: err?.message, transient: false });
|
|
1029
|
+
log(`${kind}_auth_unavailable`, { kind, reason: err?.message, ...githubFailureFields(err), transient: false });
|
|
715
1030
|
} else {
|
|
716
1031
|
authCooldownUntil.set(kind, now() + AUTH_RETRY_COOLDOWN_MS);
|
|
717
1032
|
}
|
|
@@ -758,7 +1073,7 @@ export async function startWorker(
|
|
|
758
1073
|
}
|
|
759
1074
|
};
|
|
760
1075
|
|
|
761
|
-
await attachAuth("github", makeAuth, config.github);
|
|
1076
|
+
await attachAuth("github", (cfg) => makeAuth(cfg, { log }), config.github);
|
|
762
1077
|
// GitLab joins the same map on the same best-effort terms. It appears only when configured: a forge
|
|
763
1078
|
// with no entry refuses its jobs at mint time with a message naming what is missing, which is a better
|
|
764
1079
|
// answer than an entry that exists and cannot authenticate.
|
|
@@ -844,6 +1159,10 @@ export async function startWorker(
|
|
|
844
1159
|
} catch (err) {
|
|
845
1160
|
log("log_reaper_skipped", { reason: scrubCredentials(err?.message) });
|
|
846
1161
|
}
|
|
1162
|
+
// Issue #504 part B: the allocation audit files (`allocations/YYYY-MM.jsonl`) on the same retention, which the log
|
|
1163
|
+
// reaper above never reaches (it reaps the top-level `.log` and `.json` only). Never throws, so no double wrap.
|
|
1164
|
+
const reapAllocationLogs = makeAllocationLogReaper({ logsDir: config.logsDir, retentionDays: config.logRetentionDays, log });
|
|
1165
|
+
reapAllocationLogs();
|
|
847
1166
|
|
|
848
1167
|
// REQ-RESURRECTABLE-SANDBOX: sweep retained per-job directories past their window, so what `cleanup`
|
|
849
1168
|
// kept for re-opening stays bounded. Third in the row and deliberately its own sweep -- a different
|
|
@@ -896,6 +1215,22 @@ export async function startWorker(
|
|
|
896
1215
|
// ioredis printed a stack per reconnect attempt.
|
|
897
1216
|
onValkeyError(redis, "shared client");
|
|
898
1217
|
|
|
1218
|
+
// Issue #504 part B: the applied split lives in Valkey (`alloc:plan`), shared by every host. Only with an envelope.
|
|
1219
|
+
// The boot reconcile seeds the neutral split when there is none, expires a plan past its life and re-bases on an
|
|
1220
|
+
// envelope this host changed while it was down; a fault is logged and the first pickup tries again, because the
|
|
1221
|
+
// pickup reconciles too and refuses nothing it cannot judge (it throws, and the job is retried).
|
|
1222
|
+
// Built on EVERY host, with an envelope or without: a host with none still asks, at each pickup, whether the fleet is
|
|
1223
|
+
// governed (an applied split exists), and refuses its jobs as envelope-mismatch when it is.
|
|
1224
|
+
const allocationState = makeAllocationState({ redis, host: config.workerName, audit: makeAllocationAudit({ logsDir: config.logsDir }), log });
|
|
1225
|
+
const reconcileAllocation = (why) => {
|
|
1226
|
+
if (!envelope.current) return Promise.resolve();
|
|
1227
|
+
return allocationState
|
|
1228
|
+
.reconcile({ envelope: envelope.current, digest: envelope.digest, now: new Date() })
|
|
1229
|
+
.then((r) => log("allocation_reconciled", { why, mismatch: r.mismatch, digest: envelope.digest, applied: r.state?.envelopeDigest ?? null }))
|
|
1230
|
+
.catch((err) => log("allocation_reconcile_failed", { why, reason: scrubCredentials(err?.message) }));
|
|
1231
|
+
};
|
|
1232
|
+
await reconcileAllocation("boot");
|
|
1233
|
+
|
|
899
1234
|
// This host's own stale scope claims, gated on the reaper having having enumerated: the
|
|
900
1235
|
// reaper is what establishes that this machine holds no `pi-job-*` containers, so a claim naming this
|
|
901
1236
|
// host is a claim for a container that no longer exists. Deleting it is not a second source of truth --
|
|
@@ -904,10 +1239,20 @@ export async function startWorker(
|
|
|
904
1239
|
// mechanism, so a fault costs one TTL of a stale claim and never a boot.
|
|
905
1240
|
try {
|
|
906
1241
|
if (config.workerNameDeclared)
|
|
907
|
-
await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopedLimits.current
|
|
1242
|
+
await makeScopeClaimSweeperFn({ redis, workerName: config.workerName, limits: scopeClaimRows(scopedLimits.current), log })({ reaped });
|
|
908
1243
|
} catch (err) {
|
|
909
1244
|
log("scope_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
|
|
910
1245
|
}
|
|
1246
|
+
// Issue #503: this host's stale model endpoint claims, on the same precondition and for the same reason (each is a
|
|
1247
|
+
// claim for a container). Every index up to the parse ceiling, not up to today's `slots`: `slots` can be lowered
|
|
1248
|
+
// live, and a claim on an index above the new value is still this host's to clear. A per-machine endpoint (a host
|
|
1249
|
+
// alias name) takes no fleet claim, so it has none to sweep. The sweep stops at its first fault.
|
|
1250
|
+
try {
|
|
1251
|
+
if (config.workerNameDeclared && modelEndpoints.current.length > 0)
|
|
1252
|
+
await makeClaimSweeperFn({ redis, workerName: config.workerName, keyFor: endpointSlotKey, rows: modelEndpoints.current.filter((e) => !isPerMachineHost(e.host)).map((e) => ({ hash: hash16(e.id), count: MAX_SLOTS })), event: "endpoint_claims", log })({ reaped });
|
|
1253
|
+
} catch (err) {
|
|
1254
|
+
log("endpoint_claims_sweep_skipped", { reason: scrubCredentials(err?.message) });
|
|
1255
|
+
}
|
|
911
1256
|
|
|
912
1257
|
// The persistent runtime queue: the stall guard tears schedulers down through it, AND the outbox
|
|
913
1258
|
// collector enqueues chained children onto it -- the same pi-jobs queue, so one handle serves both.
|
|
@@ -974,12 +1319,17 @@ export async function startWorker(
|
|
|
974
1319
|
// would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
|
|
975
1320
|
// no job then issues a single extra Valkey command.
|
|
976
1321
|
const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
|
|
977
|
-
const recordRun = ({ job, result, error, startedAt, endedAt }) => {
|
|
1322
|
+
const recordRun = ({ job, result, error, startedAt, endedAt, project }) => {
|
|
1323
|
+
// The project (issue #499) was resolved at the pickup gate and rides here as `project` (an id or null), so a live
|
|
1324
|
+
// edit of projects.json mid-run cannot make the record disagree with what the job was counted against. A record
|
|
1325
|
+
// path that ends BEFORE the pickup gate (the wait gate's refusals) passes none, and resolves from the live ref
|
|
1326
|
+
// with the same function.
|
|
1327
|
+
const projectId = project !== undefined ? project : projectOf(job?.data ?? {}, projects.current);
|
|
978
1328
|
// The `host` is stamped HERE rather than inside the processor, which is what keeps every one of its
|
|
979
1329
|
// four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
|
|
980
1330
|
// The default venue rides the same way and for the same reason (#277): it is the value the registry
|
|
981
1331
|
// below is built with, so the record resolves a job's venue exactly as dispatch does.
|
|
982
|
-
const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend });
|
|
1332
|
+
const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId });
|
|
983
1333
|
writeRecord(record);
|
|
984
1334
|
// STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
|
|
985
1335
|
// leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
|
|
@@ -997,25 +1347,30 @@ export async function startWorker(
|
|
|
997
1347
|
// the disarm -- the same chosen direction, met at shutdown instead of a crash.
|
|
998
1348
|
void disarmOnce({ job, endedAt });
|
|
999
1349
|
};
|
|
1350
|
+
// The record read back, for a job the queue lost the lock of after it finished (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK,
|
|
1351
|
+
// CONST-RETRY-INFRA-ONLY): this host's own file first, then the fleet's copy where a mirror is armed, because the
|
|
1352
|
+
// host that meets the stalled job need not be the one that ran it. A single host has no mirror and needs none.
|
|
1353
|
+
const settledRecord = makeSettledRecord({
|
|
1354
|
+
readRecord: makeReadRecord({ logsDir: config.logsDir }),
|
|
1355
|
+
readMirrored: runMirror ? (jobId) => readMirroredRecordFn(redis, sanitizeJobId(jobId)) : null,
|
|
1356
|
+
});
|
|
1000
1357
|
|
|
1001
1358
|
// INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
|
|
1002
|
-
// closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the
|
|
1359
|
+
// closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the fourteen
|
|
1003
1360
|
// effective settings from the overlay over env; an invalid overlay returns `{ invalid }` (logged loudly,
|
|
1004
1361
|
// key-name-only per no-pii-in-logs) so the processor RETURNS a settings-overlay-invalid refusal instead
|
|
1005
1362
|
// of the run.
|
|
1006
1363
|
const settingsFile = config.settingsFile;
|
|
1007
1364
|
const getSettings = () => {
|
|
1008
|
-
|
|
1365
|
+
// `resolveSettings` merges overlay over env and then checks the cross-key dollar rule on the MERGED values
|
|
1366
|
+
// (issue #501), so an overlay window with an env per-job cap is valid. `secretProfiles` rides alongside the
|
|
1367
|
+
// effective keys; that function says why.
|
|
1368
|
+
const res = resolveSettings(config, readOverlay(settingsFile, { log }));
|
|
1009
1369
|
if (res.invalid) {
|
|
1010
1370
|
log("settings_overlay_invalid", { reason: res.invalid, settingsFile });
|
|
1011
1371
|
return { invalid: res.invalid };
|
|
1012
1372
|
}
|
|
1013
|
-
|
|
1014
|
-
// resolves `overlay > env` over a fixed ten-key literal, and its own tests pin that key set and
|
|
1015
|
-
// assert an empty overlay returns the config verbatim -- so an eleventh key there would break both,
|
|
1016
|
-
// and would also claim a precedence this key deliberately does not have (a name declared in both
|
|
1017
|
-
// sources is refused per delivery, not silently won by either).
|
|
1018
|
-
return { ...effectiveSettings(config, res.overlay), secretProfiles: res.overlay?.secretProfiles ?? {} };
|
|
1373
|
+
return res;
|
|
1019
1374
|
};
|
|
1020
1375
|
|
|
1021
1376
|
// Resolve the Worker constructor's slot count once from the overlay: a present overlay may raise or lower
|
|
@@ -1023,6 +1378,13 @@ export async function startWorker(
|
|
|
1023
1378
|
// the per-job path enforce the refusal (getSettings already logged the invalid reason).
|
|
1024
1379
|
const bootSettings = getSettings();
|
|
1025
1380
|
const bootConcurrency = bootSettings.invalid ? config.concurrency : bootSettings.concurrency;
|
|
1381
|
+
// PR #549's review: the merged per-job cap, for the scoped-limits dollar-row warning at boot and at each reload.
|
|
1382
|
+
// An invalid overlay falls back to the env value, the same fallback as the slot count above.
|
|
1383
|
+
const deploymentMaxCostUsd = () => {
|
|
1384
|
+
const s = getSettings();
|
|
1385
|
+
return s.invalid ? config.maxCostUsd : s.maxCostUsd;
|
|
1386
|
+
};
|
|
1387
|
+
warnDollarRowsWithoutCap(scopedLimits.current, bootSettings.invalid ? config.maxCostUsd : bootSettings.maxCostUsd, log);
|
|
1026
1388
|
|
|
1027
1389
|
// INT-OUTBOX-CONTRACT chain collector: the host-side reader of a completed local parent's /outbox. It
|
|
1028
1390
|
// enqueues chained children onto the CRON queue via enqueueLocalJob -- this host's own when one is
|
|
@@ -1033,6 +1395,18 @@ export async function startWorker(
|
|
|
1033
1395
|
// else would enqueue a job only this host can run onto a queue every host drains.
|
|
1034
1396
|
const collectChain = makeCollectChain({ queue: cronQueue, config, log });
|
|
1035
1397
|
|
|
1398
|
+
// Issue #505. One live-file check, shared by the pickup gate, the snapshot and the plan collector, so the three ask one
|
|
1399
|
+
// file by one rule: the same file and rule as the one-shot checks below (`onceTriggersFile`). `pi-dispatch run
|
|
1400
|
+
// --trigger` reads it there too, so the command that fires a trigger and the check that confirms its flag see one file.
|
|
1401
|
+
const checkPortfolioFlag = makeCheckPortfolioFlag({ triggersPath: onceTriggersFile });
|
|
1402
|
+
const governingNow = () => (envelope.current ? { envelope: envelope.current, digest: envelope.digest } : null);
|
|
1403
|
+
// The plan collector: a completed portfolio job's /outbox/priorities.json, applied under the envelope by the same
|
|
1404
|
+
// allocation state every pickup reconciles. Never throws, like collectChain beside it.
|
|
1405
|
+
const collectPlan = makeCollectPlan({ allocation: allocationState, governing: governingNow, projects: () => projects.current, checkPortfolioFlag, log });
|
|
1406
|
+
// The snapshot a confirmed portfolio job reads as /job/portfolio.json, built at prepare. The run counts are complete
|
|
1407
|
+
// only with the run mirror, which a declared worker name arms (the `runMirror` rule below).
|
|
1408
|
+
const portfolioSnapshot = makePortfolioSnapshot({ checkPortfolioFlag, governing: governingNow, projects: () => projects.current, limits: () => scopedLimits.current, allocation: allocationState, redis, logsDir: config.logsDir, mirror: config.workerNameDeclared === true, log });
|
|
1409
|
+
|
|
1036
1410
|
// REQ-GLOBAL-PI-OVERLAY staged packages: read the operator's stage manifest at EACH job start, like
|
|
1037
1411
|
// getSettings above and the pause-window ref below.
|
|
1038
1412
|
//
|
|
@@ -1225,6 +1599,24 @@ export async function startWorker(
|
|
|
1225
1599
|
// here is no opinion at all, and such a host must never be able to disagree with one that has one.
|
|
1226
1600
|
fpCron: () => cronFingerprint(authoredCron(config), { tz: hostTz }) ?? "",
|
|
1227
1601
|
cronCount: () => schedules.current.length,
|
|
1602
|
+
// Issue #501 part 6: a fingerprint of the dollar caps this host judges the SHARED dollar counters against (the four
|
|
1603
|
+
// settings as a job resolves them, and the scoped-limits dollar rows), so doctor can name two hosts that would
|
|
1604
|
+
// admit different jobs against one counter. A thunk for `fpCron`'s reason: the overlay and the scoped-limits file
|
|
1605
|
+
// change without a restart. Read without a log, so an invalid overlay is not logged on every beat (each job
|
|
1606
|
+
// logs it already). With an invalid overlay it hashes the env values, the slot count's fallback above; such a
|
|
1607
|
+
// host refuses every job (settings-overlay-invalid) until the file is fixed. The env list rides along: it decides
|
|
1608
|
+
// which model rows a job without its own list reserves in.
|
|
1609
|
+
fpUsd: () => {
|
|
1610
|
+
const settings = resolveSettings(config, readOverlay(settingsFile));
|
|
1611
|
+
return usdFingerprint(settings.invalid ? config : settings, scopedLimits.current, config.allowedModels);
|
|
1612
|
+
},
|
|
1613
|
+
// Issue #499 part C: a fingerprint of the LIVE projects (ids and member hashes, never a name), so doctor can name a
|
|
1614
|
+
// host whose projects.json differs. Each host resolves its own jobs' project from its own copy, while the project
|
|
1615
|
+
// rows' counters are shared, so two copies put one repo in two projects. A thunk, so a live edit shows in one beat.
|
|
1616
|
+
fpProjects: () => projectsFingerprint(projects.current),
|
|
1617
|
+
// Issue #504 part B: the digest of this host's live envelope (`envelopeDigest`, 16 hex, never a value), so doctor can
|
|
1618
|
+
// name a host whose envelope differs; such a host refuses governed jobs as `envelope-mismatch`. `none` without one.
|
|
1619
|
+
fpEnvelope: () => envelope.digest ?? NO_ENVELOPE_FINGERPRINT,
|
|
1228
1620
|
});
|
|
1229
1621
|
|
|
1230
1622
|
|
|
@@ -1506,6 +1898,9 @@ export async function startWorker(
|
|
|
1506
1898
|
// `operator-cancel`, because the operator initiated it and a push telling them what they just did is
|
|
1507
1899
|
// noise with a pager attached.
|
|
1508
1900
|
const HOOK_POLICY_REASONS = new Set(["worker-abort", "runner-policy", ...RUNNER_POLICY_REASONS]);
|
|
1901
|
+
// One predicate for the completed listener and the lost-lock path below, so a record replays exactly the page its
|
|
1902
|
+
// result would have sent.
|
|
1903
|
+
const pagesAsPolicy = (result) => Boolean(onFailure) && result?.outcome === "policy" && HOOK_POLICY_REASONS.has(result.reason) && result.budgetReserved !== false;
|
|
1509
1904
|
// The infra-terminal sentence (issue #288). FIXED, never err.message: the message classes that reach
|
|
1510
1905
|
// a failedReason carry host paths and library words (the #310 record), and for a local job this text
|
|
1511
1906
|
// lands verbatim in the service log through the adapter's stdout fallthrough. The worker log already
|
|
@@ -1542,6 +1937,7 @@ export async function startWorker(
|
|
|
1542
1937
|
getSettings,
|
|
1543
1938
|
redis,
|
|
1544
1939
|
recordRun,
|
|
1940
|
+
settledRecord,
|
|
1545
1941
|
extraClosers,
|
|
1546
1942
|
// REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
|
|
1547
1943
|
// Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
|
|
@@ -1549,6 +1945,12 @@ export async function startWorker(
|
|
|
1549
1945
|
// Issue #242: the scoped-limits snapshot the pickup gate and the scoped budget read, once per
|
|
1550
1946
|
// pickup, from the live-reloaded ref -- same next-job grain as pauseUntil above.
|
|
1551
1947
|
scopedLimits: () => scopedLimits.current,
|
|
1948
|
+
// Issue #499: the projects snapshot, read by the pickup gate once, beside the limits snapshot above.
|
|
1949
|
+
projects: () => projects.current,
|
|
1950
|
+
// Issue #504 part B: the live envelope and its digest, and the reconcile the pickup runs before it narrows a job's
|
|
1951
|
+
// dollar ledgers by the applied split. Without an envelope `current()` is null, and `fleetGoverned` asks whether an
|
|
1952
|
+
// applied split exists: if it does, this host's jobs refuse as envelope-mismatch rather than run ungoverned.
|
|
1953
|
+
allocation: { current: () => (envelope.current ? { envelope: envelope.current, digest: envelope.digest } : null), reconcile: allocationState.reconcile, fleetGoverned: allocationState.fleetGoverned },
|
|
1552
1954
|
// Issue #230. The `after` ceiling is read per pickup from config rather than frozen into the
|
|
1553
1955
|
// processor, so it is one value with one home; the wait state shares the budget's redis client
|
|
1554
1956
|
// because it describes the same delayed jobs that client already reasons about.
|
|
@@ -1567,6 +1969,14 @@ export async function startWorker(
|
|
|
1567
1969
|
// the operator limited to one. Nothing refreshes this claim, deliberately: a refresher would be a
|
|
1568
1970
|
// second thing to get wrong for a window that cannot be reached.
|
|
1569
1971
|
scopeLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: scopeSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
|
|
1972
|
+
// Issue #503: the fleet-wide half of a model endpoint's `slots`, armed like the scope lease (declaring a name is
|
|
1973
|
+
// declaring a fleet) and with its TTL for its reason: the claim lives as long as the job's container, which
|
|
1974
|
+
// `JOB_TIMEOUT_MS` bounds. Without a declared name only the in-process bound applies, per host.
|
|
1975
|
+
endpointLease: hostQueue ? makeFleetLease({ redis, holderPrefix: config.workerName, keyFor: endpointSlotKey, ttlMs: SCOPE_CLAIM_TTL_MS, log }) : null,
|
|
1976
|
+
// The endpoint gate's snapshot, read per pickup: the live-reloaded declaration and the overlay models.json, which
|
|
1977
|
+
// the operator edits without a restart too. Read only when an endpoint is declared (the gate's own rule).
|
|
1978
|
+
modelEndpoints: () => modelEndpoints.current,
|
|
1979
|
+
overlayModels: () => readOverlayModels(config.globalPiDir),
|
|
1570
1980
|
checkLease: hostQueue
|
|
1571
1981
|
? makeFleetLease({
|
|
1572
1982
|
redis,
|
|
@@ -1584,6 +1994,7 @@ export async function startWorker(
|
|
|
1584
1994
|
maxFaults: () => config.waitMaxFaults,
|
|
1585
1995
|
deps: {
|
|
1586
1996
|
collectChain,
|
|
1997
|
+
collectPlan,
|
|
1587
1998
|
// The one-shot pre-spend check (issue #231): reads the same file the disarm writes, refuses
|
|
1588
1999
|
// only on a FOREIGN positive mark (index.mjs binds the real queue jobId so a retry of the
|
|
1589
2000
|
// spending delivery is excused). In the compose topology this check is the once-enforcement
|
|
@@ -1600,24 +2011,41 @@ export async function startWorker(
|
|
|
1600
2011
|
//
|
|
1601
2012
|
// A PROBE: whatever it resolves is dropped on the floor. The credential itself is read where it always
|
|
1602
2013
|
// was, inside buildContainerEnv, so no live key is ever in scope in the processor.
|
|
1603
|
-
|
|
2014
|
+
// Issue #503: `modelEndpoints` is the pickup's snapshot, the same one runContainer hands buildContainerEnv.
|
|
2015
|
+
checkProviderCredential: (job, { modelEndpoints = null } = {}) => {
|
|
1604
2016
|
try {
|
|
1605
|
-
resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv });
|
|
2017
|
+
resolveProviderCredential({ provider: job.provider, hostEnv: env, authFromPi: config.authFromPi, forwardEnv: config.forwardEnv, modelEndpoints });
|
|
1606
2018
|
return { ok: true };
|
|
1607
2019
|
} catch (error) {
|
|
1608
2020
|
// Only OUR determinate refusal. Anything else (a bug here, an fs fault the module does not model)
|
|
1609
2021
|
// must not become a policy refusal on the operator's issue: it rethrows into runJob's catch, which
|
|
1610
2022
|
// classifies it the way it always did.
|
|
2023
|
+
// Issue #503: a transient overlay read at this pickup is no verdict; the processor retries it as infra.
|
|
2024
|
+
if (error?.piDispatchTransient === true) return { ok: false, unavailable: error.code ?? "unreadable" };
|
|
1611
2025
|
if (error?.piDispatchConfig !== true) throw error;
|
|
1612
2026
|
return { ok: false, message: error.message };
|
|
1613
2027
|
}
|
|
1614
2028
|
},
|
|
2029
|
+
// Issue #502. The model-exists gate: pi's builtin catalog, then the overlay models.json, read at most once per
|
|
2030
|
+
// job and only when a model is not builtin, so the operator's edits apply without a restart. The SAME file
|
|
2031
|
+
// the endpoint gate reads, through the same reader, so absent, unreadable and unparseable mean one thing.
|
|
2032
|
+
checkModelsKnown: (refs) => checkModelsKnown(refs, { readOverlay: () => readOverlayModels(config.globalPiDir) }),
|
|
2033
|
+
// Issue #503 part 7: the builtin catalog's model object, for the zero-rated check that lets a job on local
|
|
2034
|
+
// zero-rated models reserve nothing in the dollar windows (processor.mjs, `zeroRatedVerdict`).
|
|
2035
|
+
builtinModel,
|
|
2036
|
+
// Issue #502. The deployment's allowed-model list (PI_ALLOWED_MODELS), null = unrestricted. Env only, and
|
|
2037
|
+
// handed to the processor here rather than through the settings overlay, which a model-callable tool writes.
|
|
2038
|
+
// index.mjs folds it into the effective job (`effectiveJobOf`) under the trigger's own `run.models`.
|
|
2039
|
+
allowedModels: config.allowedModels,
|
|
1615
2040
|
// Issue #230. The same file and the same fail-open posture, but its own mtime-cached read: this one
|
|
1616
2041
|
// asks whether the AUTHORED entry declares wait conditions the job arrived without, which is how a
|
|
1617
2042
|
// service below the version floor turns a wait into a paid run nothing can tell from a correct
|
|
1618
2043
|
// one. In the compose topology the worker's read is the live inode while the receiver's is dead
|
|
1619
2044
|
// until restart, which is exactly the deployment where the skew happens.
|
|
1620
2045
|
checkWaitSkew: makeCheckWaitSkew({ triggersPath: onceTriggersFile }),
|
|
2046
|
+
// Issue #505. Whether the live file still flags a portfolio job's cron trigger, read when such a job is picked
|
|
2047
|
+
// up, from the same file and by the same rule as the two checks above (built once, beside collectPlan).
|
|
2048
|
+
checkPortfolioFlag,
|
|
1621
2049
|
// Issue #230. Whether a job the supersede lease names is still in the queue. Without it a holder
|
|
1622
2050
|
// that vanished by any route except the clean one leaves a key that refuses every later delivery
|
|
1623
2051
|
// for that target until it expires -- and a refused forge delivery is gone, since no webhook
|
|
@@ -1694,6 +2122,11 @@ export async function startWorker(
|
|
|
1694
2122
|
jobImage: config.jobImage,
|
|
1695
2123
|
// #277: the venue a retained directory records, which the sandbox refuses by when it is not here.
|
|
1696
2124
|
defaultBackend: config.defaultBackend,
|
|
2125
|
+
// Issue #504 part B: a local job's folder is resolved at prepare and mounted resolved; one named inside a run
|
|
2126
|
+
// root or a cron folder must still resolve inside it, and none may be the envelope's folder or above it.
|
|
2127
|
+
localPlacement: { jobPaths: envelopeJobPathsNow, envelopeFile: config.envelopeFile ?? null },
|
|
2128
|
+
// Issue #505: /job/portfolio.json for a confirmed portfolio job.
|
|
2129
|
+
portfolioSnapshot,
|
|
1697
2130
|
preparers: makeForgePreparers({ gitlabApiUrl: config.gitlab?.apiUrl ?? null, forgejoApiUrl: config.forgejo?.apiUrl ?? null, azureOrgUrl: config.azure?.orgUrl ?? null }),
|
|
1698
2131
|
// The cron event.json's previousRunAt (INT-CONTAINER-JOB-INPUTS): read back from the same
|
|
1699
2132
|
// per-job run-history sidecars recordRun writes above -- no new store, no new query surface.
|
|
@@ -1762,7 +2195,7 @@ export async function startWorker(
|
|
|
1762
2195
|
// comment and the signal for CONST-PI-VERSION-PINNED's silent-no-op mode -- a missing line is
|
|
1763
2196
|
// what tells a human a run did nothing. The container's own output already streams via
|
|
1764
2197
|
// runContainer's onOutput during the run.
|
|
1765
|
-
// `reason` is a fixed enum (worker-abort | over-budget | unprotected-branch | runner-policy |
|
|
2198
|
+
// `reason` is a fixed enum (worker-abort | over-budget | dollar-cap | allocation-cap | envelope-mismatch | portfolio-no-envelope | portfolio-snapshot-oversize | unprotected-branch | runner-policy |
|
|
1766
2199
|
// provider-auth-refused | job-image-missing), never
|
|
1767
2200
|
// user content. Included only when present so success lines stay clean; a shutdown-aborted job logs
|
|
1768
2201
|
// { outcome: "policy", reason: "worker-abort" }, making a restart-dropped job visible.
|
|
@@ -1775,11 +2208,16 @@ export async function startWorker(
|
|
|
1775
2208
|
// and never in the failed listener -- a failed-only mount would miss exactly the paid terminals
|
|
1776
2209
|
// the feature exists for. Folded into the existing listener body, never a second w.on: the
|
|
1777
2210
|
// start-wiring harness records ONE handler per event, and two would race the log line's pin.
|
|
1778
|
-
|
|
2211
|
+
// `budgetReserved !== false` (issue #502): `model-not-allowed` is both a runner stop (paid, pages) and a
|
|
2212
|
+
// pre-spend refusal of a main model outside the job's list (free, comments, pages nobody), and the reason
|
|
2213
|
+
// alone cannot tell them apart. Every paid terminal above carries `budgetReserved: true`.
|
|
2214
|
+
if (pagesAsPolicy(result)) {
|
|
1779
2215
|
onFailure({ jobId: job?.id, outcome: "policy", reason: result.reason });
|
|
1780
2216
|
}
|
|
1781
2217
|
});
|
|
1782
|
-
for
|
|
2218
|
+
// The failed listener's body, for a failure the queue decided. Split out only so the lost-lock check below can
|
|
2219
|
+
// run it after an await; a failure of any other reason runs it synchronously, exactly as before.
|
|
2220
|
+
const onFailed = (job, err) => {
|
|
1783
2221
|
log("job_failed", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? err).slice(0, 120) });
|
|
1784
2222
|
// The one reason whose sentence is the fix itself (issue #458): logged WHOLE, beside the cut line above.
|
|
1785
2223
|
if (err?.reason === NETNS_KEEPER_NOT_HOLDING || err?.reason === NETNS_KEEPER_CRASH_LOOP) log("job_failed_netns_keeper", { jobId: job?.id, attempt: job?.attemptsMade, reason: String(err?.message ?? "") });
|
|
@@ -1794,6 +2232,29 @@ export async function startWorker(
|
|
|
1794
2232
|
void comment({ ...job.data, id: job.id }, (typeof err?.reason === "string" && Object.hasOwn(FAILED_COMMENT_BY_REASON, err.reason) ? FAILED_COMMENT_BY_REASON[err.reason] : null) ?? FAILED_COMMENT);
|
|
1795
2233
|
onFailure?.({ jobId: job?.id, outcome: "failed", reason: typeof err?.reason === "string" ? err.reason : "infra" });
|
|
1796
2234
|
}
|
|
2235
|
+
};
|
|
2236
|
+
// A job that FINISHED, then lost its lock (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK). When Valkey is unreachable
|
|
2237
|
+
// for longer than the lock renewal window while a processor finishes, the record is written, BullMQ refuses the
|
|
2238
|
+
// completion ("Missing lock"), and its stall check (`maxStalledCount: 0`) fails the job at the next pickup with
|
|
2239
|
+
// STALLED_FAILED_REASON. Without this check that job posted the failure comment and paged the operator for a run
|
|
2240
|
+
// that ended. So a terminal stall failure first asks the run record: when the record says this attempt finished
|
|
2241
|
+
// without failing, the job's terminal line is `job_lost_lock_after_completion` and nothing is posted. The page
|
|
2242
|
+
// the completed listener would have sent for that record (a paid policy stop) is sent here instead, because
|
|
2243
|
+
// that listener never saw the first finish. No record, an earlier attempt's, a `failed` one, or a lookup fault
|
|
2244
|
+
// keeps the failure path below: the comment and the page.
|
|
2245
|
+
for (const w of allWorkers) w.on("failed", (job, err) => {
|
|
2246
|
+
if (!(job?.finishedOn && err?.message === STALLED_FAILED_REASON)) return onFailed(job, err);
|
|
2247
|
+
void (async () => {
|
|
2248
|
+
// `attemptsMade` is read AFTER BullMQ's moveToFailed added one, so it equals the attempt number the
|
|
2249
|
+
// record carries (`buildRecord`'s `attemptsMade + 1`, written while the job was processing).
|
|
2250
|
+
// A record found and refused is said, with its fixed reason, so an operator reading the failure comment
|
|
2251
|
+
// can see why the record did not suppress it (a clock skew past the tolerance reads `older-than-job`).
|
|
2252
|
+
const onReject = (reason, source) => log("job_lost_lock_record_rejected", { jobId: job.id, reason, source });
|
|
2253
|
+
const record = await settledRecord(job.id, { attempt: job.attemptsMade, since: job.timestamp, onReject });
|
|
2254
|
+
if (!record) return onFailed(job, err);
|
|
2255
|
+
log("job_lost_lock_after_completion", { jobId: job.id, outcome: record.outcome, ...(record.reason ? { reason: record.reason } : {}) });
|
|
2256
|
+
if (pagesAsPolicy(record)) onFailure({ jobId: job.id, outcome: "policy", reason: record.reason });
|
|
2257
|
+
})().catch(() => {}); // `settledRecord` never rejects; this only keeps a throwing log line from going unhandled
|
|
1797
2258
|
});
|
|
1798
2259
|
|
|
1799
2260
|
// CONST-RETRY-INFRA-ONLY money backstop: BullMQ's maxStalledCount does not bound scheduler jobs, so a
|
|
@@ -1844,7 +2305,7 @@ export async function startWorker(
|
|
|
1844
2305
|
// Only when a triggers file is configured; best-effort; a bad edit keeps the running schedulers. The
|
|
1845
2306
|
// closer each of the three returns joins `extraClosers`, so the watch stops with the worker (issue #295).
|
|
1846
2307
|
if (config.triggersFile) {
|
|
1847
|
-
extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers));
|
|
2308
|
+
extraClosers.push(watchTriggersFile(config, cronQueue, log, schedules, registry, hostTz, config.workerNameDeclared, atBoot.triggers, checkEnvelopePlace));
|
|
1848
2309
|
}
|
|
1849
2310
|
|
|
1850
2311
|
// REQ-SCOPED-PAUSE-WINDOWS live edit: watch the pause-windows file and hot-swap the in-memory windows, so
|
|
@@ -1853,9 +2314,29 @@ export async function startWorker(
|
|
|
1853
2314
|
extraClosers.push(watchPauseWindowsFile(config, pauseWindows, log, atBoot.pauseWindows));
|
|
1854
2315
|
}
|
|
1855
2316
|
|
|
2317
|
+
// Issue #499 part B: the two files a project row joins reload as a pair; the deployment cap rides along so a limits
|
|
2318
|
+
// list either reload commits gets the dollar-rows-without-cap warning.
|
|
2319
|
+
const projectPair = makeProjectPair(scopedLimits, projects, deploymentMaxCostUsd);
|
|
2320
|
+
// Issue #504 part B: the envelope reloads with the pair, since its floors are judged against both files: a committed
|
|
2321
|
+
// limits or projects edit re-judges it from disk, and its own edits are judged against the live pair.
|
|
2322
|
+
const envelopeCtx = { projects, limits: scopedLimits, maxCostMicros: envelopeCap, jobPaths: envelopeJobPathsNow, onChange: () => void reconcileAllocation("reload") };
|
|
2323
|
+
if (config.envelopeFile) {
|
|
2324
|
+
projectPair.afterCommit = () => reloadEnvelope(config, envelope, log, envelopeCtx);
|
|
2325
|
+
extraClosers.push(watchEnvelopeFn(config, envelope, log, atBoot.envelope, envelopeCtx));
|
|
2326
|
+
}
|
|
1856
2327
|
// Issue #242 live edit: hot-swap the scoped limits on file change, keeping last-good on a bad edit.
|
|
1857
2328
|
if (config.scopedLimitsFile) {
|
|
1858
|
-
extraClosers.push(
|
|
2329
|
+
extraClosers.push(watchScopedLimitsFn(config, scopedLimits, log, atBoot.scopedLimits, deploymentMaxCostUsd, projectPair));
|
|
2330
|
+
}
|
|
2331
|
+
|
|
2332
|
+
// Issue #499 live edit: the projects file, keep-last-good on a bad edit.
|
|
2333
|
+
if (config.projectsFile) {
|
|
2334
|
+
extraClosers.push(watchProjectsFn(config, projects, log, atBoot.projects, projectPair));
|
|
2335
|
+
}
|
|
2336
|
+
|
|
2337
|
+
// Issue #503 live edit: the model endpoints, keep-last-good on a bad edit.
|
|
2338
|
+
if (watchModelEndpoints) {
|
|
2339
|
+
extraClosers.push(watchModelEndpointsFile(config, modelEndpoints, log, atBoot.modelEndpoints));
|
|
1859
2340
|
}
|
|
1860
2341
|
|
|
1861
2342
|
// issue #292 / OQ-007: re-run the three retention sweeps on a timer, because the supported deployment
|
|
@@ -1873,6 +2354,8 @@ export async function startWorker(
|
|
|
1873
2354
|
const sweep = makeRetentionSweepFn({
|
|
1874
2355
|
reapers: [
|
|
1875
2356
|
{ name: "log", reap: reapLogs },
|
|
2357
|
+
// Issue #504 part B: where boot runs it, right after the run history it sits beside.
|
|
2358
|
+
{ name: "allocation_log", reap: reapAllocationLogs },
|
|
1876
2359
|
{ name: "sandbox", reap: reapSandboxes },
|
|
1877
2360
|
{ name: "session", reap: () => sessionStore.reapSessions() },
|
|
1878
2361
|
],
|
|
@@ -1912,6 +2395,11 @@ export async function startWorker(
|
|
|
1912
2395
|
softHoldPct: config.softHoldPct, // null when the soft-hold band is disabled
|
|
1913
2396
|
scopedLimitsFile: config.scopedLimitsFile, // null = no scoped caps/concurrency (the folder mutex holds regardless)
|
|
1914
2397
|
scopedLimits: scopedLimits.current.length, // row count -- money config deserves boot visibility; the watcher logs only changes
|
|
2398
|
+
projectsFile: config.projectsFile, // issue #499: null = no projects
|
|
2399
|
+
projects: projects.current.length, // project count, never a name
|
|
2400
|
+
envelopeFile: config.envelopeFile, // issue #504: null = no envelope and no delegation
|
|
2401
|
+
envelopeDigest: envelope.digest, // the envelope's 16-hex digest, the host row's fpEnvelope; null without one
|
|
2402
|
+
modelEndpoints: modelEndpoints.current.length, // issue #503: declared model endpoints, each a slot lease at pickup
|
|
1915
2403
|
image: config.jobImage,
|
|
1916
2404
|
valkey: config.valkeyUrl,
|
|
1917
2405
|
// Issue #464: the literal address every Valkey client of this worker dials, beside the URL as written; null
|