@edgehero/pi-dispatch 4.0.1 → 4.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -1
- package/src/capacity-cli.mjs +375 -0
- package/src/capacity-records.mjs +209 -0
- package/src/capacity.mjs +654 -0
- package/src/cli.mjs +9 -0
- package/src/config.mjs +4 -20
- package/src/connection.mjs +4 -2
- package/src/doctor.mjs +214 -36
- package/src/host-budget.mjs +6 -0
- package/src/host-registry.mjs +20 -2
- package/src/index.mjs +97 -6
- package/src/job-size.mjs +16 -0
- package/src/live-jobs.mjs +135 -0
- package/src/prepare.mjs +1 -9
- package/src/repeat-slot.mjs +13 -0
- package/src/run-earlier.mjs +29 -0
- package/src/run-history.mjs +115 -5
- package/src/run-mirror.mjs +161 -12
- package/src/size-suggest.mjs +34 -14
- package/src/start.mjs +39 -4
- package/src/wait-for.mjs +7 -0
- package/src/worker-name.mjs +25 -0
package/src/index.mjs
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { availableParallelism } from "node:os";
|
|
1
2
|
import { DelayedError, UnrecoverableError, Worker } from "bullmq";
|
|
2
3
|
import { assertJudgedConnection, onValkeyError } from "./connection.mjs";
|
|
3
4
|
import { jobContainerName } from "./backend-local.mjs";
|
|
@@ -7,7 +8,7 @@ import { CANCEL_ACK_TTL_MS, cancelAckKey, cancelReqKey } from "./cancel-state.mj
|
|
|
7
8
|
import { InfraRetry, NETNS_KEEPER_CRASH_LOOP, NETNS_KEEPER_NOT_HOLDING, TERMINAL_COMMENTS, runJob } from "./processor.mjs";
|
|
8
9
|
import { NETNS_KEEPER_YOUNG_HOLD_MAX_MS, netnsKeeperCrashLoopSentence, netnsKeeperLoopAgainSentence } from "./netns-keeper.mjs";
|
|
9
10
|
import { PODMAN_RESTART_HOLD_EXPIRED, PODMAN_RESTART_HOLD_MAX_MS, PODMAN_RESTART_HOLD_RECHECK_MS } from "./runtime-observations.mjs";
|
|
10
|
-
import { targetFor } from "./run-history.mjs";
|
|
11
|
+
import { earlierFrom, targetFor } from "./run-history.mjs";
|
|
11
12
|
import { isPerMachineHost } from "./backends.mjs";
|
|
12
13
|
import { hash16 } from "./fleet-lease.mjs";
|
|
13
14
|
import { endpointsForModel } from "./model-endpoints.mjs";
|
|
@@ -190,7 +191,43 @@ export function effectiveJobOf(data, settings, allowedModels = null, log = () =>
|
|
|
190
191
|
};
|
|
191
192
|
}
|
|
192
193
|
|
|
193
|
-
|
|
194
|
+
/**
|
|
195
|
+
* The jobs this host runs right now (issue #599, phase 2): one entry per PICKUP that holds a slot, `{ id, project, memMiB,
|
|
196
|
+
* cpuCenti, at }`, `at` its admission instant in millis (the record's `startedAt`). The registry beat publishes it
|
|
197
|
+
* (`jobs`, live-jobs.mjs), and the capacity report counts each as busy until its record exists.
|
|
198
|
+
*
|
|
199
|
+
* It works WITHOUT a host budget, which is why it is not the budget's ledger: the ledger exists only when a budget is
|
|
200
|
+
* set. Keyed by a per-pickup number, not the job id, so a second pickup of the same id (a stalled job handed back while
|
|
201
|
+
* the first still runs here) can never remove the first's entry. Process memory, `makeInFlight`'s reason: it counts
|
|
202
|
+
* this process's own pickups, and a restart that loses it has no pickup left to count.
|
|
203
|
+
*
|
|
204
|
+
* An entry is added where the job is admitted (`startedAt`, `capacity`) and removed only by `releaseAllHolds`, the one
|
|
205
|
+
* release every exit goes through, so it cannot outlive its pickup. An ORPHAN (the container's stop did not take) is
|
|
206
|
+
* removed like any other exit: the processor is done with it, and only the host budget keeps watching its container,
|
|
207
|
+
* so the beat flags orphans from the budget's ledger alone. Without a budget nothing watches such a container, and the
|
|
208
|
+
* list says only what this process still runs.
|
|
209
|
+
*/
|
|
210
|
+
export function makeRunningJobs() {
|
|
211
|
+
const entries = new Map();
|
|
212
|
+
let next = 0;
|
|
213
|
+
return {
|
|
214
|
+
/** Adds one pickup's entry; returns its key. Total: it never throws, so the admission block may call it. */
|
|
215
|
+
add(entry) {
|
|
216
|
+
next += 1;
|
|
217
|
+
entries.set(next, { ...entry });
|
|
218
|
+
return next;
|
|
219
|
+
},
|
|
220
|
+
/** Removes one pickup's entry; true only the first time. */
|
|
221
|
+
remove(key) {
|
|
222
|
+
return entries.delete(key);
|
|
223
|
+
},
|
|
224
|
+
/** Copies, never the live map. */
|
|
225
|
+
list: () => [...entries.values()].map((e) => ({ ...e })),
|
|
226
|
+
size: () => entries.size,
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
export function makeProcessor({ cancelJob, stopContainer, containerName = (job) => jobContainerName(job.id), redis, getSettings, hostBudget = null, abortGraceMs = ABORT_GRACE_MS, applyConcurrency = () => {}, pauseUntil = () => null, scopedLimits = () => [], projects = () => [], jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), hostBound = null, checkLease = null, scopeLease = null, endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels = () => null, endpointSetFor = mainModelEndpoints, deps, recordRun = () => {}, settledRecord = null, previousRecord = null, timeoutMs = JOB_TIMEOUT_MS, cancelPollMs = 2_000, cancelStopBoundMs = CANCEL_STOP_BOUND_MS, hostName = "", multiHost = false, now = () => Date.now(), waitState = makeWaitState({ redis, now }), afterMaxMs = () => WAIT_AFTER_MAX_DEFAULT_MS, checkSlots = makeInFlight(), checkSlotCount = () => 1, checkTimeoutMs = () => 10_000, concurrencyNow = () => 3, intervalMs = () => WAIT_INTERVAL_FLOOR_MS * 2, maxWaitMs = () => 24 * 3600 * 1000, maxChecks = () => 96, maxFaults = () => 5, random = Math.random, cpus = availableParallelism, runningJobs = makeRunningJobs() }) {
|
|
194
231
|
// The lost-lock gate's rejection lines, said ONCE per job id, stall count and attempt. The gate runs before every
|
|
195
232
|
// deferral (pause window, wait, scope or endpoint busy), and BullMQ never resets a job's stall count, so a stalled
|
|
196
233
|
// scheduled job whose record is refused meets the gate again on every deferred pickup: a 30-minute scope-busy hold
|
|
@@ -233,6 +270,19 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
233
270
|
}
|
|
234
271
|
}
|
|
235
272
|
|
|
273
|
+
// THE EARLIER ATTEMPTS' SLOT TIME (issue #599): a retry, or a pickup after a stall, writes its record over the one
|
|
274
|
+
// before it (one file and one mirror key per job id), so that record is read NOW, before anything here can replace
|
|
275
|
+
// it, and its slot interval (with the ones it carried) rides every record this pickup writes (`earlier`). Only on
|
|
276
|
+
// such a pickup, so a first attempt reads nothing; bounded by the reader, and a fault reads as none.
|
|
277
|
+
let earlier = null;
|
|
278
|
+
if (typeof previousRecord === "function" && ((Number.isInteger(job.attemptsMade) && job.attemptsMade > 0) || Number(job.stalledCounter) > 0)) {
|
|
279
|
+
try {
|
|
280
|
+
earlier = earlierFrom(await previousRecord(job.id));
|
|
281
|
+
} catch {
|
|
282
|
+
earlier = null;
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
|
|
236
286
|
// Scoped pause windows (REQ-SCOPED-PAUSE-WINDOWS): if this job's folder/repo is inside an active pause
|
|
237
287
|
// window, DEFER it to the window end via BullMQ's delayed set -- the job keeps its identity/dedup and
|
|
238
288
|
// auto-resumes when re-picked. This is FIRST, before the kill timer, the settings read, and the budget
|
|
@@ -302,7 +352,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
302
352
|
const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false, hostBudget: { memMiB: budgetNow.memMiB, cpuCenti: budgetNow.cpuCenti, hostShare: share } };
|
|
303
353
|
// Above the wait gate, so through the recorder's own arguments rather than the bound one below it: the
|
|
304
354
|
// same pickup project and size, by hand once.
|
|
305
|
-
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString(), project, size });
|
|
355
|
+
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString(), project, size, earlier });
|
|
306
356
|
return result;
|
|
307
357
|
}
|
|
308
358
|
deps?.log?.("job_size_never_fits_here_deferred", { jobId: job.id, project, misfit, delayMs: NEVER_FITS_RECHECK_MS, ...sizeFields });
|
|
@@ -344,7 +394,7 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
344
394
|
// resolver path or a vault topology -- `secret-profile-unknown` sets that rule.
|
|
345
395
|
if (sentence && deps?.comment) await Promise.resolve(deps.comment(job.data, sentence)).catch(() => {});
|
|
346
396
|
const result = { outcome: "policy", reason, exitCode: null, turns: null, tokens: null, budgetReserved: false };
|
|
347
|
-
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString() });
|
|
397
|
+
recordRun({ job, result, startedAt: at, endedAt: new Date().toISOString(), earlier });
|
|
348
398
|
return result;
|
|
349
399
|
};
|
|
350
400
|
|
|
@@ -673,6 +723,8 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
673
723
|
// guard) still frees every local slot before it rethrows; the fleet halves are release-if-mine and awaited where
|
|
674
724
|
// the caller can. Draining, so a second call releases nothing: the in-process map's release is not idempotent.
|
|
675
725
|
let budgetHeld = false;
|
|
726
|
+
// This pickup's entry in the running jobs (issue #599, phase 2), null until it is admitted below.
|
|
727
|
+
let runningKey = null;
|
|
676
728
|
let stopDidNotTake = false;
|
|
677
729
|
let name;
|
|
678
730
|
let venue;
|
|
@@ -692,6 +744,11 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
692
744
|
return Promise.all(taken.map((hold) => hold.fleet?.release?.()));
|
|
693
745
|
};
|
|
694
746
|
const releaseAllHolds = ({ orphan = false } = {}) => {
|
|
747
|
+
// The running-jobs entry first (issue #599, phase 2): this pickup no longer runs the job, orphan or not.
|
|
748
|
+
if (runningKey !== null) {
|
|
749
|
+
runningJobs.remove(runningKey);
|
|
750
|
+
runningKey = null;
|
|
751
|
+
}
|
|
695
752
|
const endpointsReleased = releaseEndpointHolds();
|
|
696
753
|
const scopesReleased = releaseScopeHolds();
|
|
697
754
|
if (hostHeld) {
|
|
@@ -718,11 +775,31 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
718
775
|
hostHeld = true;
|
|
719
776
|
}
|
|
720
777
|
|
|
778
|
+
// The host's capacity as the record carries it (issue #599). Read when the job is admitted, not here: the slot count,
|
|
779
|
+
// the budget and the runtime's CPU count are live, and the value a run is judged against is the one it started under.
|
|
780
|
+
const capacityNow = () => {
|
|
781
|
+
const read = (fn) => {
|
|
782
|
+
try {
|
|
783
|
+
return fn();
|
|
784
|
+
} catch {
|
|
785
|
+
return null;
|
|
786
|
+
}
|
|
787
|
+
};
|
|
788
|
+
const budget = hostBudget ? read(() => hostBudget.current()) : null;
|
|
789
|
+
// The CPUs are the RUNTIME's count where its facts were read (the budget's own read), the worker's only when
|
|
790
|
+
// they were not: on Docker Desktop the worker sees the Mac's cores while every job runs in the VM's.
|
|
791
|
+
const runtimeCpus = hostBudget ? read(() => hostBudget.hostCpus?.()) : null;
|
|
792
|
+
return { slots: read(concurrencyNow), memMiB: budget?.memMiB ?? null, cpuCenti: budget?.cpuCenti ?? null, cpus: Number.isSafeInteger(runtimeCpus) && runtimeCpus >= 1 ? runtimeCpus : read(cpus) };
|
|
793
|
+
};
|
|
721
794
|
// THE ONE RECORDER BELOW THE GATE, bound once, so the pickup project is a property of the path and not of each call
|
|
722
795
|
// site: every record from here on goes through it, and none can drop the field and fall back to the live ref in
|
|
723
796
|
// start.mjs, which would disagree with the pickup value exactly when projects.json was edited mid-run. A bolt in
|
|
724
797
|
// project-pickup.test.mjs refuses a bare `recordRun(` call below this line.
|
|
725
|
-
|
|
798
|
+
//
|
|
799
|
+
// `capacity` (issue #599) rides the same binding, and is null until the job is admitted below (where `startedAt` is
|
|
800
|
+
// set): every record of a refusal before that point says, by its null, that the job never held a slot.
|
|
801
|
+
let capacity = null;
|
|
802
|
+
const recordAfterGate = (args) => recordRun({ ...args, project, size, capacity, earlier });
|
|
726
803
|
|
|
727
804
|
// The MATCHED ROW's scope keys both the in-process slot and the fleet lease (issue #498), the same string
|
|
728
805
|
// `budgetCapsFor` hashes below and the boot sweeper hashes from the file: a qualified `github:acme/web` row holds
|
|
@@ -984,6 +1061,14 @@ export function makeProcessor({ cancelJob, stopContainer, containerName = (job)
|
|
|
984
1061
|
// exception, and it is named: the registry's refusal of an unheld venue is caught at the call, and
|
|
985
1062
|
// anything else it throws propagates to the catch below, which releases what was acquired.
|
|
986
1063
|
startedAt = new Date().toISOString();
|
|
1064
|
+
// What this host offered as the job took its slot (issue #599, INT-RUN-HISTORY-FILE-CONTRACT): the live slot
|
|
1065
|
+
// count, the host budget (null without one; `recordedCapacity` writes its `Infinity` as "off") and the CPUs the OS
|
|
1066
|
+
// reports. Each read guarded: this block must not throw (above), and a fact that cannot be read is unknown, never
|
|
1067
|
+
// a reason to fail a job that is about to run.
|
|
1068
|
+
capacity = capacityNow();
|
|
1069
|
+
// This host runs the job from here on (issue #599, phase 2): its entry, at the same instant as `startedAt`, given
|
|
1070
|
+
// back by `releaseAllHolds` on every exit. `add` is total, so this block still cannot throw.
|
|
1071
|
+
runningKey = runningJobs.add({ id: job.id, project, memMiB: size.memMiB, cpuCenti: size.cpuCenti, at: Date.parse(startedAt) });
|
|
987
1072
|
// The producer of the name both boot reapers sweep by substring. Built from the shared prefix
|
|
988
1073
|
// rather than typed here, so a rename cannot land in the producer and not in the sweeps (#227).
|
|
989
1074
|
// From the VENUE that will build the container, not from the local adapter reached for directly:
|
|
@@ -1517,7 +1602,7 @@ export function keeperHoldState(job, at) {
|
|
|
1517
1602
|
return { at, since, startedMs };
|
|
1518
1603
|
}
|
|
1519
1604
|
|
|
1520
|
-
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, limiter, pauseUntil, scopedLimits, projects, jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [], hostBudget: hostBudgetOptions = null }) {
|
|
1605
|
+
export function createWorker({ connection, name, stopContainer, containerName, hostQueue = null, checkLease = null, scopeLease = null, checkTimeoutMs, concurrency, getSettings, redis, deps, recordRun, settledRecord = null, previousRecord = null, limiter, pauseUntil, scopedLimits, projects, jobSizeEnv = {}, allocation = null, inFlight = makeInFlight(), waitState, afterMaxMs, checkSlots = makeInFlight(), checkSlotCount, concurrencyNow, intervalMs, maxWaitMs, maxChecks, maxFaults, hostSlots = makeInFlight(), endpointSlots = makeInFlight(), endpointLease = null, modelEndpoints = null, overlayModels, extraClosers = [], hostBudget: hostBudgetOptions = null, runningJobs = makeRunningJobs() }) {
|
|
1521
1606
|
// One Worker per queue name (issue #57). A host-affine job -- one whose folder, secret resolver or wait
|
|
1522
1607
|
// check lives on THIS machine -- is enqueued to `pi-jobs@<name>` rather than filtered for at pickup,
|
|
1523
1608
|
// because BullMQ has no selective pop and the put-it-back alternative does not work: promotion out of
|
|
@@ -1626,6 +1711,10 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1626
1711
|
// The run-record lookup a stalled job is checked against before it can run again (see the processor's
|
|
1627
1712
|
// first gate). `null` in a bare wiring, which keeps today's behaviour: the job runs.
|
|
1628
1713
|
settledRecord,
|
|
1714
|
+
previousRecord,
|
|
1715
|
+
// Issue #599, phase 2: SHARED across both workers for the host slot's reason: the registry row lists what this
|
|
1716
|
+
// HOST runs, and two maps would each list half.
|
|
1717
|
+
runningJobs,
|
|
1629
1718
|
});
|
|
1630
1719
|
|
|
1631
1720
|
// Issue #464: only a connection `parseConnection` built, which judges and pins the Valkey it dials.
|
|
@@ -1652,6 +1741,8 @@ export function createWorker({ connection, name, stopContainer, containerName, h
|
|
|
1652
1741
|
primary.hostWorker = workers[1] ?? null;
|
|
1653
1742
|
// Issue #596, phase 2: the host budget, for the registry beat's thunks (start.mjs) and doctor-facing snapshots.
|
|
1654
1743
|
primary.hostBudget = hostBudget;
|
|
1744
|
+
// Issue #599, phase 2: the jobs this host runs now, for the registry beat's `jobs` thunk (start.mjs).
|
|
1745
|
+
primary.runningJobs = runningJobs;
|
|
1655
1746
|
|
|
1656
1747
|
const stop = async () => {
|
|
1657
1748
|
// Abort active jobs (=> docker stop via onAbort), then close. Without the cancel,
|
package/src/job-size.mjs
CHANGED
|
@@ -54,6 +54,16 @@ export const JOB_MEMORY_CEILING_MIB = 1024 * 1024;
|
|
|
54
54
|
export const JOB_CPUS_FLOOR_CENTI = 25;
|
|
55
55
|
/** The largest CPU size: 256, as hundredths, because round(256 x 1024) is 262144, the top of `--cpu-shares`. */
|
|
56
56
|
export const JOB_CPUS_CEILING_CENTI = 256 * 100;
|
|
57
|
+
/**
|
|
58
|
+
* The most a host can have, as a run record's `capacity` may say it (issue #599, phase 4's review): a value past these is
|
|
59
|
+
* not a host but a fault or a forgery, so the writer records it as unknown and the capacity report counts a record that
|
|
60
|
+
* says it as unreadable, rather than judging a week against "1 of 1000000000 slots". The CPUs and the memory are the
|
|
61
|
+
* ceilings the runtime facts are read to (daemon-facts.mjs `cpuCount`, `memoryMiB`); the slots are as many of the
|
|
62
|
+
* smallest jobs (`JOB_CPUS_FLOOR_CENTI`) as those CPUs hold.
|
|
63
|
+
*/
|
|
64
|
+
export const HOST_CPUS_MAX = 4096;
|
|
65
|
+
export const HOST_MEMORY_MAX_MIB = 64 * 1024 * 1024;
|
|
66
|
+
export const HOST_SLOTS_MAX = (HOST_CPUS_MAX * 100) / JOB_CPUS_FLOOR_CENTI;
|
|
57
67
|
/** The valid range of `--cpu-shares` on cgroup v2 (runc and crun clamp to it; 2 is the kernel's cgroup v1 minimum). */
|
|
58
68
|
export const CPU_SHARES_MIN = 2;
|
|
59
69
|
export const CPU_SHARES_MAX = 262144;
|
|
@@ -119,6 +129,12 @@ export function parseCpus(value) {
|
|
|
119
129
|
return centi;
|
|
120
130
|
}
|
|
121
131
|
|
|
132
|
+
/**
|
|
133
|
+
* The two never-fits refusals (index.mjs `SIZE_REFUSAL_COMMENTS`, issue #596), both decided before the job holds a slot.
|
|
134
|
+
* Here, beside the size they judge, so a leaf can read them; a test holds them equal to the processor's.
|
|
135
|
+
*/
|
|
136
|
+
export const SIZE_REFUSAL_REASONS = Object.freeze(["job-size-exceeds-host", "job-size-exceeds-share"]);
|
|
137
|
+
|
|
122
138
|
/**
|
|
123
139
|
* The one spelling of a memory size: `<n>g` when it is whole gigabytes, else `<n>m`. Every `--memory`, `--memory-swap`
|
|
124
140
|
* and stored row is written by this function, and `container-spec.mjs`'s `memoryBytes` reads every value it can write
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The running jobs a host row publishes (issue #599, phase 2, INT-HOST-REGISTRY-CONTRACT `jobs`), written by the
|
|
3
|
+
* worker's beat and read back by every reader of the row: one module for both halves, so the writer can never publish
|
|
4
|
+
* a shape the reader drops, and pure (no clock, no filesystem, no Valkey), so the capacity report and the admin bundle
|
|
5
|
+
* can import it.
|
|
6
|
+
*
|
|
7
|
+
* WHAT IS LISTED. Every job this host's processors hold a slot for right now (the in-flight map, `makeRunningJobs` in
|
|
8
|
+
* index.mjs), and every ORPHAN of the host budget (a job whose container's stop did not take, or a container left
|
|
9
|
+
* from before the worker started), flagged `o: 1`. An orphan comes from the budget's ledger only, because only the
|
|
10
|
+
* budget keeps watching such a container until the runtime says it is gone; without a budget nothing does, and the
|
|
11
|
+
* job leaves the list when its processor ends. A job listed by both is listed once, as running.
|
|
12
|
+
*
|
|
13
|
+
* WHY THIS MEETS THE CONTENT RULE (names, integers and digests). Every field is an integer, the fixed 1, or an id:
|
|
14
|
+
* - `id` is the job id, the one every run record already carries in `jobId` and `runs:rec:<id>` already puts in
|
|
15
|
+
* Valkey. Each shape the project mints is charset-checked by construction: `gh-`, `gl-`, `fj-` and the other forge
|
|
16
|
+
* prefixes with the forge's delivery id (and `-r<n>` for a replica), `repeat:<trigger id>:<millis>` and
|
|
17
|
+
* `manual:<trigger id>:<millis>` (a trigger id is `[A-Za-z0-9._-]+`, the loader's rule), `local-<hex>`,
|
|
18
|
+
* `chain-<hex>`, and a budget orphan's `container:<name>`. A delivery id is a forge's header value and is not
|
|
19
|
+
* checked by the receiver, so the id is held to `LIVE_JOB_ID_RE` HERE, and one outside it (or longer than 128
|
|
20
|
+
* characters) is published as its digest (`sha256:<16 hex>`), the rule's own idiom for a value that must be
|
|
21
|
+
* carried and cannot satisfy it. No quote, backslash, slash or control character can reach the JSON, so the
|
|
22
|
+
* writer's path-shape check never refuses the field.
|
|
23
|
+
* - `p` is a project id (`isProjectId`) or null; `runs:rec:*` already carries project ids.
|
|
24
|
+
* - `m`, `c` and `at` are integers (MiB, hundredths of a CPU, epoch millis), `o` is 1.
|
|
25
|
+
*
|
|
26
|
+
* ORDER AND SIZE. Oldest first (by `at`, then id), at most `LIVE_JOBS_MAX`: the jobs that have held a slot longest
|
|
27
|
+
* carry the most busy time and are the ones an operator looks for. The rest are counted in `jobsMore`, never dropped
|
|
28
|
+
* silently. A full list is about 7 KiB, once per beat.
|
|
29
|
+
*/
|
|
30
|
+
|
|
31
|
+
import { createHash } from "node:crypto";
|
|
32
|
+
import { isProjectId } from "./project-id.mjs";
|
|
33
|
+
|
|
34
|
+
/** The most jobs one row lists; the rest are counted (`jobsMore`). */
|
|
35
|
+
export const LIVE_JOBS_MAX = 32;
|
|
36
|
+
/** A job id as published: the charset every id shape the project mints is in, at most 128 characters. */
|
|
37
|
+
export const LIVE_JOB_ID_RE = /^[A-Za-z0-9._:-]{1,128}$/;
|
|
38
|
+
/** The most a `jobs` value may be before a reader declines to parse it: 32 entries at their longest fit well inside. */
|
|
39
|
+
export const LIVE_JOBS_MAX_BYTES = 16 * 1024;
|
|
40
|
+
|
|
41
|
+
const positiveInt = (v) => Number.isSafeInteger(v) && v > 0;
|
|
42
|
+
|
|
43
|
+
/** A job id as it may be published: itself when inside `LIVE_JOB_ID_RE`, else its digest; null for no id. */
|
|
44
|
+
export function publishedJobId(id) {
|
|
45
|
+
if (typeof id !== "string" || id === "") return null;
|
|
46
|
+
if (LIVE_JOB_ID_RE.test(id)) return id;
|
|
47
|
+
return `sha256:${createHash("sha256").update(id).digest("hex").slice(0, 16)}`;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* The list a row publishes, before the cap: `[{ id, p, m, c, at, o? }]`, oldest first. `running` is the in-flight
|
|
52
|
+
* map's entries (`{ id, project, memMiB, cpuCenti, at }`), `budgetEntries` the host budget's ledger (`entries()`), of
|
|
53
|
+
* which only orphans are taken. An entry with no id or no instant is left out and counted in `skipped`.
|
|
54
|
+
*/
|
|
55
|
+
export function liveJobsOf({ running = [], budgetEntries = [] } = {}) {
|
|
56
|
+
const out = [];
|
|
57
|
+
const seen = new Set();
|
|
58
|
+
let skipped = 0;
|
|
59
|
+
const push = (e, orphan) => {
|
|
60
|
+
const id = publishedJobId(e?.id);
|
|
61
|
+
if (id === null || !Number.isSafeInteger(e?.at) || e.at <= 0) {
|
|
62
|
+
skipped++;
|
|
63
|
+
return;
|
|
64
|
+
}
|
|
65
|
+
if (seen.has(id)) return; // one job, two pickups (or running and orphaned): listed once
|
|
66
|
+
seen.add(id);
|
|
67
|
+
out.push({ id, p: isProjectId(e.project) ? e.project : null, ...(positiveInt(e.memMiB) ? { m: e.memMiB } : {}), ...(positiveInt(e.cpuCenti) ? { c: e.cpuCenti } : {}), at: e.at, ...(orphan ? { o: 1 } : {}) });
|
|
68
|
+
};
|
|
69
|
+
// Running first, so a job both running and in the ledger as an orphan (it cannot be: the gate defers it) reads as running.
|
|
70
|
+
for (const e of Array.isArray(running) ? running : []) push(e, false);
|
|
71
|
+
for (const e of Array.isArray(budgetEntries) ? budgetEntries : []) if (e?.orphan) push(e, true);
|
|
72
|
+
out.sort((a, b) => a.at - b.at || (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
|
|
73
|
+
return { jobs: out, skipped };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* The two row fields: `jobs`, the JSON of at most `LIVE_JOBS_MAX` entries, and `jobsMore`, how many running jobs are
|
|
78
|
+
* not in it (past the cap, or with no id or instant), both strings as a row holds them.
|
|
79
|
+
*/
|
|
80
|
+
export function liveJobsFields(list) {
|
|
81
|
+
const { jobs, skipped } = list;
|
|
82
|
+
const kept = jobs.slice(0, LIVE_JOBS_MAX);
|
|
83
|
+
return { jobs: JSON.stringify(kept), jobsMore: String(jobs.length - kept.length + skipped) };
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* A row's `jobs` read back, through a per-field allowlist: `{ jobs, dropped }`, `jobs` an array of `{ id, p, m, c, at,
|
|
88
|
+
* o }` (`m`, `c` null when not published, `o` true for an orphan), or null when the field is absent, empty, too long
|
|
89
|
+
* or not a JSON list. An entry that is not an object, or whose id, project, size, instant or flag is not the shape
|
|
90
|
+
* above, is dropped and counted; past `LIVE_JOBS_MAX` entries the rest are counted too. A list this function already
|
|
91
|
+
* read (an array, its `o` a boolean) reads back the same, so a caller may hand either form. Never throws: a row is
|
|
92
|
+
* another host's text.
|
|
93
|
+
*/
|
|
94
|
+
export function parseLiveJobs(raw) {
|
|
95
|
+
if (Array.isArray(raw)) return parseList(raw);
|
|
96
|
+
if (typeof raw !== "string" || raw === "" || raw.length > LIVE_JOBS_MAX_BYTES) return { jobs: null, dropped: 0 };
|
|
97
|
+
let parsed;
|
|
98
|
+
try {
|
|
99
|
+
parsed = JSON.parse(raw);
|
|
100
|
+
} catch {
|
|
101
|
+
return { jobs: null, dropped: 0 };
|
|
102
|
+
}
|
|
103
|
+
return Array.isArray(parsed) ? parseList(parsed) : { jobs: null, dropped: 0 };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const optionalSize = (v) => (v === undefined || v === null ? null : positiveInt(v) ? v : undefined);
|
|
107
|
+
|
|
108
|
+
function parseList(list) {
|
|
109
|
+
const jobs = [];
|
|
110
|
+
let dropped = 0;
|
|
111
|
+
for (const e of list) {
|
|
112
|
+
if (jobs.length >= LIVE_JOBS_MAX) {
|
|
113
|
+
dropped++;
|
|
114
|
+
continue;
|
|
115
|
+
}
|
|
116
|
+
const ok = e !== null && typeof e === "object" && !Array.isArray(e);
|
|
117
|
+
const m = ok ? optionalSize(e.m) : undefined;
|
|
118
|
+
const c = ok ? optionalSize(e.c) : undefined;
|
|
119
|
+
if (!ok || typeof e.id !== "string" || !LIVE_JOB_ID_RE.test(e.id) || !(e.p === undefined || e.p === null || isProjectId(e.p)) || m === undefined || c === undefined || !positiveInt(e.at) || !(e.o === undefined || e.o === 1 || typeof e.o === "boolean")) {
|
|
120
|
+
dropped++;
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
jobs.push({ id: e.id, p: e.p ?? null, m, c, at: e.at, o: e.o === 1 || e.o === true });
|
|
124
|
+
}
|
|
125
|
+
return { jobs, dropped };
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* A row's `jobsMore` read back: a non-negative integer, or null when absent or not one. A reader adds `dropped` from
|
|
130
|
+
* `parseLiveJobs` to it: a listed job it could not read is a running job it does not count.
|
|
131
|
+
*/
|
|
132
|
+
export function parseJobsMore(raw) {
|
|
133
|
+
if (Number.isSafeInteger(raw) && raw >= 0) return raw;
|
|
134
|
+
return typeof raw === "string" && /^\d{1,9}$/.test(raw) ? Number(raw) : null;
|
|
135
|
+
}
|
package/src/prepare.mjs
CHANGED
|
@@ -15,6 +15,7 @@ import { buildAzurePrompt } from "./azure-prompt.mjs";
|
|
|
15
15
|
import { prepareLocalWorkspace } from "./prepare-local.mjs";
|
|
16
16
|
import { copySkillTree } from "./copy-tree.mjs";
|
|
17
17
|
import { recordedJobSize } from "./job-size.mjs";
|
|
18
|
+
import { scheduledForMillis } from "./repeat-slot.mjs";
|
|
18
19
|
|
|
19
20
|
/**
|
|
20
21
|
* The subdirectory of the per-job dir a trigger's injected skills are copied into, so they reach the
|
|
@@ -214,15 +215,6 @@ function localEventContext(job, queueJobId, findPreviousRun) {
|
|
|
214
215
|
return { source: "manual" };
|
|
215
216
|
}
|
|
216
217
|
|
|
217
|
-
/** Parse the millis out of a `repeat:<id>:<millis>` BullMQ scheduled jobId, or null. */
|
|
218
|
-
function scheduledForMillis(queueJobId) {
|
|
219
|
-
if (typeof queueJobId !== "string" || !queueJobId.startsWith("repeat:")) return null;
|
|
220
|
-
const tail = queueJobId.slice(queueJobId.lastIndexOf(":") + 1);
|
|
221
|
-
if (tail === "") return null;
|
|
222
|
-
const millis = Number(tail);
|
|
223
|
-
return Number.isFinite(millis) ? millis : null;
|
|
224
|
-
}
|
|
225
|
-
|
|
226
218
|
/**
|
|
227
219
|
* Attach the sandbox stamp to a successful prepare, and to nothing else.
|
|
228
220
|
*
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The scheduled-for instant of a BullMQ job scheduler's job, read from the deterministic `repeat:<id>:<millis>` id it
|
|
3
|
+
* mints (DES-CRON-VIA-BULLMQ-SCHEDULER). A leaf module, so the run record and the prepare step read one parser.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/** Parse the millis out of a `repeat:<id>:<millis>` BullMQ scheduled jobId, or null. */
|
|
7
|
+
export function scheduledForMillis(queueJobId) {
|
|
8
|
+
if (typeof queueJobId !== "string" || !queueJobId.startsWith("repeat:")) return null;
|
|
9
|
+
const tail = queueJobId.slice(queueJobId.lastIndexOf(":") + 1);
|
|
10
|
+
if (tail === "") return null;
|
|
11
|
+
const millis = Number(tail);
|
|
12
|
+
return Number.isFinite(millis) ? millis : null;
|
|
13
|
+
}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The rule for a run record's `earlier` (issue #599, INT-RUN-HISTORY-FILE-CONTRACT), in a module that imports only the
|
|
3
|
+
* worker name rule, so the record writer (run-history.mjs, which re-exports both names) and the capacity report, a pure
|
|
4
|
+
* reader, rebuild it with ONE validator.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { WORKER_NAME_RE } from "./worker-name.mjs";
|
|
8
|
+
|
|
9
|
+
const recordInt = (v) => (Number.isSafeInteger(v) && v >= 0 ? v : null);
|
|
10
|
+
|
|
11
|
+
/** The most earlier attempts a record carries: the newest are kept. */
|
|
12
|
+
export const EARLIER_MAX = 4;
|
|
13
|
+
|
|
14
|
+
/** One earlier attempt as a record carries it, or null: a worker name, two canonical ISO instants in order, two sizes. */
|
|
15
|
+
function earlierEntry(e) {
|
|
16
|
+
if (e === null || typeof e !== "object" || Array.isArray(e)) return null;
|
|
17
|
+
const instant = (v) => (typeof v === "string" && Number.isFinite(Date.parse(v)) && new Date(Date.parse(v)).toISOString() === v ? v : null);
|
|
18
|
+
const startedAt = instant(e.startedAt);
|
|
19
|
+
const endedAt = instant(e.endedAt);
|
|
20
|
+
if (typeof e.host !== "string" || !WORKER_NAME_RE.test(e.host) || startedAt === null || endedAt === null || Date.parse(endedAt) < Date.parse(startedAt)) return null;
|
|
21
|
+
return { host: e.host, startedAt, endedAt, memMiB: recordInt(e.memMiB), cpuCenti: recordInt(e.cpuCenti) };
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** A record's `earlier`, rebuilt: the valid entries, the newest `EARLIER_MAX` of them, or null when none is. */
|
|
25
|
+
export function recordedEarlier(list) {
|
|
26
|
+
if (!Array.isArray(list)) return null;
|
|
27
|
+
const kept = list.map(earlierEntry).filter((e) => e !== null).slice(-EARLIER_MAX);
|
|
28
|
+
return kept.length > 0 ? kept : null;
|
|
29
|
+
}
|
package/src/run-history.mjs
CHANGED
|
@@ -6,7 +6,12 @@ import { resolveBackendName } from "./backend-registry.mjs";
|
|
|
6
6
|
import { isForgeKind, targetSeparator } from "./forges.mjs";
|
|
7
7
|
import { MODEL_REF_PATTERN as USAGE_ID_PATTERN } from "./model-ref.mjs";
|
|
8
8
|
import { isProjectId } from "./project-id.mjs";
|
|
9
|
-
import { recordedJobSize } from "./job-size.mjs";
|
|
9
|
+
import { HOST_CPUS_MAX, HOST_MEMORY_MAX_MIB, HOST_SLOTS_MAX, recordedJobSize } from "./job-size.mjs";
|
|
10
|
+
import { waitArmed } from "./wait-for.mjs";
|
|
11
|
+
import { scheduledForMillis } from "./repeat-slot.mjs";
|
|
12
|
+
import { EARLIER_MAX, recordedEarlier } from "./run-earlier.mjs";
|
|
13
|
+
|
|
14
|
+
export { EARLIER_MAX, recordedEarlier };
|
|
10
15
|
|
|
11
16
|
/**
|
|
12
17
|
* Durable per-run history.
|
|
@@ -569,7 +574,7 @@ function rebuildUsage(u) {
|
|
|
569
574
|
* default to `null` when the outcome does not carry them, so the record shape is stable whether or not
|
|
570
575
|
* the source reports those fields.
|
|
571
576
|
*/
|
|
572
|
-
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null, size = null }) {
|
|
577
|
+
export function buildRecord({ job, result, error, startedAt, endedAt, host = null, defaultBackend = null, project = null, size = null, capacity = null, earlier = null }) {
|
|
573
578
|
const data = job.data ?? {};
|
|
574
579
|
const kind = data.kind ?? job.name;
|
|
575
580
|
const source = result ?? error ?? {};
|
|
@@ -728,7 +733,114 @@ export function buildRecord({ job, result, error, startedAt, endedAt, host = nul
|
|
|
728
733
|
// refused: an off or unknown dimension never refuses by itself); `hostShare` an integer or null. Present only on
|
|
729
734
|
// the `job-size-exceeds-host` and `-share` records; null on every other.
|
|
730
735
|
hostBudget: recordedHostBudget(source.hostBudget),
|
|
736
|
+
// When the job became ELIGIBLE to run (issue #599, INT-RUN-HISTORY-FILE-CONTRACT), so `startedAt - queuedAt` is how
|
|
737
|
+
// long it waited for a slot. Additive, nullable, an explicit literal, TAIL position after `hostBudget` on the same
|
|
738
|
+
// contract. `job.timestamp + opts.delay`, not `job.timestamp` alone: BullMQ's job scheduler creates the next cron
|
|
739
|
+
// job when the current one runs, stamped with that moment and delayed until it is due (job-scheduler.js
|
|
740
|
+
// `getNextJobOpts`), so the timestamp alone made a daily trigger read as a day of waiting. `opts.delay` is the
|
|
741
|
+
// delay the job was ADDED with and survives a `moveToDelayed` (a pause, a deferral) and a retry, while `job.delay`
|
|
742
|
+
// is rewritten by both (measured against bullmq 5.80.4 on Valkey, worker/test/queued-at.integration.test.mjs), so
|
|
743
|
+
// a pause window and a deferral count as waiting, which they are. Null for a job held on `run.waitFor` (a wait its
|
|
744
|
+
// trigger asked for is not a wait for capacity) and for a retry (its wait would include the earlier attempt). An
|
|
745
|
+
// ISO string or null, a number's rendering, so PII-free.
|
|
746
|
+
queuedAt: queuedAtOf(job),
|
|
747
|
+
// What this host offered when the job took its slot (issue #599, INT-RUN-HISTORY-FILE-CONTRACT): `{ slots, memMiB,
|
|
748
|
+
// cpuCenti, cpus }`, the live PI_CONCURRENCY, the host budget as `hostBudget` above maps it, and the CPUs the OS
|
|
749
|
+
// reports. Additive, nullable, an explicit literal REBUILT here (`recordedCapacity`), TAIL position after `queuedAt`.
|
|
750
|
+
// Passed in by the processor only once the job is ADMITTED (index.mjs, where `startedAt` is set), so it is also
|
|
751
|
+
// the record's own answer to "did this run hold a slot": null on every refusal before one, which is what
|
|
752
|
+
// lets a capacity report tell a run from a refusal without guessing from the wall time (capacity.mjs).
|
|
753
|
+
capacity: recordedCapacity(capacity),
|
|
754
|
+
// The slot time of this job's EARLIER attempts (issue #599, INT-RUN-HISTORY-FILE-CONTRACT), which a retry's record
|
|
755
|
+
// would otherwise erase: it overwrites the same file and the same mirror key. `[{ host, startedAt, endedAt, memMiB,
|
|
756
|
+
// cpuCenti }]`, at most `EARLIER_MAX`, newest last, read by the processor from the previous record BEFORE this one
|
|
757
|
+
// replaces it (`earlierFrom`) and passed in. Additive, nullable, an explicit literal REBUILT here, TAIL position after
|
|
758
|
+
// `capacity`. A worker name, two ISO instants and two integers per entry: never the earlier attempt's dollars,
|
|
759
|
+
// tokens or usage, so every cost reader, which reads the top-level fields only, counts nothing twice. Null on a
|
|
760
|
+
// first attempt and when the previous record held no slot or could not be read.
|
|
761
|
+
earlier: recordedEarlier(earlier),
|
|
762
|
+
// Whether the FIRST attempt stalled and this pickup re-ran it (issue #599): BullMQ raised `stalledCounter`, not
|
|
763
|
+
// `attemptsMade`, so that pickup wrote no record and its slot time is in none. A stall after a failed attempt is a
|
|
764
|
+
// retry like any other (its earlier attempt's record is carried in `earlier`), so it is false there. A boolean,
|
|
765
|
+
// TAIL position after `earlier`, so the capacity report can say how many runs that is.
|
|
766
|
+
stalledRepick: Number.isInteger(job.stalledCounter) && job.stalledCounter > 0 && !(Number.isInteger(job.attemptsMade) && job.attemptsMade > 0),
|
|
767
|
+
};
|
|
768
|
+
}
|
|
769
|
+
|
|
770
|
+
/**
|
|
771
|
+
* Of two copies of one job's record (this host's file and the fleet's copy), the one a retry replaces: the HIGHER
|
|
772
|
+
* `attempt` (a retry on another host writes its own copy there, so either side can be the newer attempt), an exact tie
|
|
773
|
+
* broken by the later `endedAt` (run-mirror.mjs `mergeRuns`' rule), a remaining tie by the first given (the local file).
|
|
774
|
+
* Either may be null; a non-object is none.
|
|
775
|
+
*/
|
|
776
|
+
export function newerRecord(a, b) {
|
|
777
|
+
const ok = (r) => r !== null && typeof r === "object" && !Array.isArray(r);
|
|
778
|
+
if (!ok(a)) return ok(b) ? b : null;
|
|
779
|
+
if (!ok(b)) return a;
|
|
780
|
+
const attempt = (r) => (Number.isInteger(r.attempt) ? r.attempt : 0);
|
|
781
|
+
if (attempt(a) !== attempt(b)) return attempt(a) > attempt(b) ? a : b;
|
|
782
|
+
const end = (r) => {
|
|
783
|
+
const t = Date.parse(r.endedAt ?? r.startedAt ?? "");
|
|
784
|
+
return Number.isFinite(t) ? t : -Infinity;
|
|
785
|
+
};
|
|
786
|
+
return end(b) > end(a) ? b : a;
|
|
787
|
+
}
|
|
788
|
+
|
|
789
|
+
/**
|
|
790
|
+
* What a retry's record carries forward from the record it is about to replace: that record's own `earlier`, then its
|
|
791
|
+
* own slot interval when it held a slot (`capacity` an object), each rebuilt; null when there is nothing.
|
|
792
|
+
*/
|
|
793
|
+
export function earlierFrom(previous) {
|
|
794
|
+
if (previous === null || typeof previous !== "object" || Array.isArray(previous)) return null;
|
|
795
|
+
const carried = Array.isArray(previous.earlier) ? previous.earlier : [];
|
|
796
|
+
const own = previous.capacity !== null && typeof previous.capacity === "object" && !Array.isArray(previous.capacity) ? [{ host: previous.host, startedAt: previous.startedAt, endedAt: previous.endedAt, memMiB: previous.size?.memMiB, cpuCenti: previous.size?.cpuCenti }] : [];
|
|
797
|
+
return recordedEarlier([...carried, ...own]);
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
/**
|
|
801
|
+
* When a job became eligible to run, as an ISO string, or null: `job.timestamp` plus the delay it was added with
|
|
802
|
+
* (`opts.delay`, at least 0). Null for a job armed with `run.waitFor` (its wait is its trigger's own), for a retry
|
|
803
|
+
* (`attemptsMade` above 0) or a pickup after a stall (`stalledCounter` above 0), and for any input that is not a
|
|
804
|
+
* finite, representable instant.
|
|
805
|
+
*/
|
|
806
|
+
export function queuedAtOf(job) {
|
|
807
|
+
if (waitArmed(job?.data)) return null;
|
|
808
|
+
// A RETRY's wait would include its earlier attempt and that attempt's run (the add's moment is kept across attempts),
|
|
809
|
+
// so only a first attempt says how long the job waited for a slot. `attemptsMade` counts attempts finished before.
|
|
810
|
+
if (Number.isInteger(job?.attemptsMade) && job.attemptsMade > 0) return null;
|
|
811
|
+
// A pickup after a STALL is the same: BullMQ raises `stalledCounter`, not `attemptsMade`, and the add's moment is kept.
|
|
812
|
+
if (Number.isInteger(job?.stalledCounter) && job.stalledCounter > 0) return null;
|
|
813
|
+
// A job scheduler's job carries its exact slot in its id. BullMQ stamps `timestamp` and computes `delay` from two
|
|
814
|
+
// separate clock reads, so their sum can land a few milliseconds past the slot (measured in CI).
|
|
815
|
+
const slot = scheduledForMillis(job?.id);
|
|
816
|
+
if (Number.isSafeInteger(slot) && slot >= 0 && slot <= 8.64e15) return new Date(slot).toISOString();
|
|
817
|
+
const timestamp = job?.timestamp;
|
|
818
|
+
const delay = job?.opts?.delay ?? 0;
|
|
819
|
+
if (!Number.isSafeInteger(timestamp) || timestamp < 0 || typeof delay !== "number" || !Number.isFinite(delay)) return null;
|
|
820
|
+
const at = timestamp + Math.max(0, Math.trunc(delay));
|
|
821
|
+
// 8.64e15 is the last instant a Date can hold; past it `toISOString` throws, and this function must not.
|
|
822
|
+
return at <= 8.64e15 ? new Date(at).toISOString() : null;
|
|
823
|
+
}
|
|
824
|
+
|
|
825
|
+
/** A record's non-negative safe integer, else null. */
|
|
826
|
+
const recordInt = (v) => (Number.isSafeInteger(v) && v >= 0 ? v : null);
|
|
827
|
+
/** A record's budget dimension: an integer, `"off"` (the budget's `Infinity`, or the word itself), else null. */
|
|
828
|
+
const recordBudget = (v) => (v === Infinity || v === "off" ? "off" : recordInt(v));
|
|
829
|
+
|
|
830
|
+
/**
|
|
831
|
+
* A run's capacity as a record carries it: `{ slots, memMiB, cpuCenti, cpus }`, else null. `slots` and `cpus` are
|
|
832
|
+
* positive integers or null (unknown); each budget dimension is `recordedHostBudget`'s mapping, so `"off"` and null
|
|
833
|
+
* keep meaning "not limited" and "not known".
|
|
834
|
+
*/
|
|
835
|
+
export function recordedCapacity(value) {
|
|
836
|
+
if (value === null || typeof value !== "object" || Array.isArray(value)) return null;
|
|
837
|
+
// Each held to what a host can have (job-size.mjs `HOST_*_MAX`): past it a value is a fault, recorded as unknown.
|
|
838
|
+
const upTo = (max, read) => (v) => {
|
|
839
|
+
const n = read(v);
|
|
840
|
+
return typeof n === "number" && n > max ? null : n;
|
|
731
841
|
};
|
|
842
|
+
const positive = (v) => (Number.isSafeInteger(v) && v >= 1 ? v : null);
|
|
843
|
+
return { slots: upTo(HOST_SLOTS_MAX, positive)(value.slots), memMiB: upTo(HOST_MEMORY_MAX_MIB, recordBudget)(value.memMiB), cpuCenti: upTo(HOST_CPUS_MAX * 100, recordBudget)(value.cpuCenti), cpus: upTo(HOST_CPUS_MAX, positive)(value.cpus) };
|
|
732
844
|
}
|
|
733
845
|
|
|
734
846
|
/**
|
|
@@ -739,9 +851,7 @@ export function buildRecord({ job, result, error, startedAt, endedAt, host = nul
|
|
|
739
851
|
*/
|
|
740
852
|
export function recordedHostBudget(value) {
|
|
741
853
|
if (value === null || typeof value !== "object" || Array.isArray(value)) return null;
|
|
742
|
-
|
|
743
|
-
const budget = (v) => (v === Infinity || v === "off" ? "off" : int(v));
|
|
744
|
-
return { memMiB: budget(value.memMiB), cpuCenti: budget(value.cpuCenti), hostShare: int(value.hostShare) };
|
|
854
|
+
return { memMiB: recordBudget(value.memMiB), cpuCenti: recordBudget(value.cpuCenti), hostShare: recordInt(value.hostShare) };
|
|
745
855
|
}
|
|
746
856
|
|
|
747
857
|
/** What became of a collected plan. */
|