@edgehero/pi-dispatch 4.0.0 → 4.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +6 -2
- package/src/capacity-cli.mjs +375 -0
- package/src/capacity-records.mjs +209 -0
- package/src/capacity.mjs +654 -0
- package/src/cli.mjs +9 -0
- package/src/config.mjs +4 -20
- package/src/connection.mjs +4 -2
- package/src/doctor.mjs +214 -36
- package/src/host-budget.mjs +6 -0
- package/src/host-registry.mjs +20 -2
- package/src/index.mjs +97 -6
- package/src/job-size.mjs +16 -0
- package/src/live-jobs.mjs +135 -0
- package/src/prepare.mjs +1 -9
- package/src/provider-steering.mjs +1 -0
- package/src/repeat-slot.mjs +13 -0
- package/src/run-earlier.mjs +29 -0
- package/src/run-history.mjs +115 -5
- package/src/run-mirror.mjs +161 -12
- package/src/size-suggest.mjs +34 -14
- package/src/start.mjs +39 -4
- package/src/wait-for.mjs +7 -0
- package/src/worker-name.mjs +25 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@edgehero/pi-dispatch",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.1.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "The pi-dispatch worker and CLI: runs the pi coding agent as a self-hosted service, one locked down Docker or Podman container per job, with spend caps checked before anything is spent, plus init, up, doctor and service install.",
|
|
6
6
|
"keywords": [
|
|
@@ -65,6 +65,10 @@
|
|
|
65
65
|
"./container-spec": "./src/container-spec.mjs",
|
|
66
66
|
"./job-size": "./src/job-size.mjs",
|
|
67
67
|
"./size-suggest": "./src/size-suggest.mjs",
|
|
68
|
+
"./capacity": "./src/capacity.mjs",
|
|
69
|
+
"./capacity-records": "./src/capacity-records.mjs",
|
|
70
|
+
"./capacity-cli": "./src/capacity-cli.mjs",
|
|
71
|
+
"./live-jobs": "./src/live-jobs.mjs",
|
|
68
72
|
"./size-records": "./src/size-records.mjs",
|
|
69
73
|
"./host-budget": "./src/host-budget.mjs",
|
|
70
74
|
"./backend-conformance": "./src/backend-conformance.mjs",
|
|
@@ -109,7 +113,7 @@
|
|
|
109
113
|
"start": "node src/cli.mjs worker"
|
|
110
114
|
},
|
|
111
115
|
"dependencies": {
|
|
112
|
-
"@earendil-works/pi-ai": "1.0
|
|
116
|
+
"@earendil-works/pi-ai": "1.1.0",
|
|
113
117
|
"@octokit/auth-app": "8.2.0",
|
|
114
118
|
"@octokit/rest": "22.0.1",
|
|
115
119
|
"bullmq": "5.80.4",
|
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `pi-dispatch capacity [--since 24h|7d|30d] [--host <name>] [--json] [--valkey-url <url>]` (issue #599,
|
|
3
|
+
* REQ-CAPACITY-INSIGHTS): how busy each host was over the window, read from the run records (`capacity-records.mjs`)
|
|
4
|
+
* and computed by the one function every surface shares (`capacity.mjs`). `--json` prints the report itself
|
|
5
|
+
* (INT-CAPACITY-REPORT).
|
|
6
|
+
*
|
|
7
|
+
* READ-ONLY, and needs what `status` needs and no more: the Valkey URL (the kill switch's rule, `killSwitchValkeyUrls`)
|
|
8
|
+
* and the deployment's logs directory, never `loadConfig`, so it answers on a deployment whose forge auth is broken.
|
|
9
|
+
* Where this shell and the deployment's `.env` name different Valkeys it refuses until `--valkey-url` says which: a
|
|
10
|
+
* report from the wrong one would read a busy fleet as idle. An unreachable or refused Valkey taken from the shell or
|
|
11
|
+
* the `.env` is not a failure: the local files are read and the report says why. One named with `--valkey-url` is: the
|
|
12
|
+
* operator asked for that Valkey's fleet, and a local-only report would answer another question.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { parseArgs } from "node:util";
|
|
16
|
+
import { CAPACITY_WINDOWS, computeCapacity } from "./capacity.mjs";
|
|
17
|
+
import { readCapacityRecords } from "./capacity-records.mjs";
|
|
18
|
+
import { formatCpus, formatMemory } from "./job-size.mjs";
|
|
19
|
+
|
|
20
|
+
/** How long the CLI waits on the host registry, the kill switch's budget. */
|
|
21
|
+
const FLEET_READ_TIMEOUT_MS = 2_000;
|
|
22
|
+
/** The deployment keys this command reads beside the Valkey URL. */
|
|
23
|
+
export const CAPACITY_ENV_KEYS = Object.freeze(["PI_LOGS_DIR", "PI_LOG_RETENTION_DAYS", "PI_WORKER_NAME"]);
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Run the command. Returns the exit code. Every collaborator is a seam with the production default.
|
|
27
|
+
* `deploymentEnv` is cli.mjs `cliDeploymentEnv`, handed in so this module does not import the CLI's entry module.
|
|
28
|
+
*/
|
|
29
|
+
export async function runCapacity(args, { env = process.env, write = (chunk) => process.stdout.write(chunk), errWrite = (chunk) => process.stderr.write(chunk), now = () => Date.now(), deploymentEnv = (e) => ({ env: e }), valkeyRefusal = async () => null, redisFn, readLiveHostsFn, fs, pickUrls } = {}) {
|
|
30
|
+
const fail = (message) => {
|
|
31
|
+
errWrite(`error: ${message}\n`);
|
|
32
|
+
return 1;
|
|
33
|
+
};
|
|
34
|
+
let parsed;
|
|
35
|
+
try {
|
|
36
|
+
parsed = parseArgs({ args, allowPositionals: false, options: { since: { type: "string", default: "7d" }, host: { type: "string" }, json: { type: "boolean", default: false }, "valkey-url": { type: "string" } } });
|
|
37
|
+
} catch (error) {
|
|
38
|
+
return fail(`${error.message}\n usage: pi-dispatch capacity [--since 24h|7d|30d] [--host <name>] [--json] [--valkey-url <url>]`);
|
|
39
|
+
}
|
|
40
|
+
const { since, host, json } = parsed.values;
|
|
41
|
+
const window = Object.hasOwn(CAPACITY_WINDOWS, since) ? CAPACITY_WINDOWS[since] : null;
|
|
42
|
+
if (!window) return fail(`--since takes ${Object.keys(CAPACITY_WINDOWS).join(", ")} (got ${JSON.stringify(since)})`);
|
|
43
|
+
|
|
44
|
+
const deployment = deploymentEnv(env, CAPACITY_ENV_KEYS);
|
|
45
|
+
if (deployment.problem) return fail(deployment.problem);
|
|
46
|
+
const facts = await deploymentFacts(deployment.env);
|
|
47
|
+
if (facts.problem) return fail(facts.problem);
|
|
48
|
+
const { logsDir, retentionDays, localHost } = facts;
|
|
49
|
+
|
|
50
|
+
const { killSwitchValkeyUrls, makeRedisClient, urlShown } = await import("./connection.mjs");
|
|
51
|
+
const picked = (pickUrls ?? killSwitchValkeyUrls)({ env, flagUrl: parsed.values["valkey-url"] ?? null });
|
|
52
|
+
if (picked.error) return fail(picked.error);
|
|
53
|
+
if (picked.note) errWrite(`warning: ${picked.note}\n`);
|
|
54
|
+
if (picked.urls.length > 1) return fail(`${picked.disagreement}: a report from the wrong one reads its fleet as idle. Say which: pi-dispatch capacity --valkey-url <url>`);
|
|
55
|
+
const url = picked.urls[0];
|
|
56
|
+
// A Valkey the operator NAMED that cannot be read is a failure: they asked for that one. One taken from the shell or
|
|
57
|
+
// the `.env` is not: the report falls back to this host's files and says why.
|
|
58
|
+
const named = parsed.values["valkey-url"] !== undefined;
|
|
59
|
+
|
|
60
|
+
const nowMs = now();
|
|
61
|
+
const windowStartMs = nowMs - window.ms;
|
|
62
|
+
let redis = null;
|
|
63
|
+
let refusedNote = null;
|
|
64
|
+
const refused = await valkeyRefusal(url, env);
|
|
65
|
+
if (refused && named) return fail(`Valkey at ${urlShown(url)} refused: ${refused}`);
|
|
66
|
+
if (refused) refusedNote = `Valkey at ${urlShown(url)} refused (${refused}): only this host's files were read`;
|
|
67
|
+
else {
|
|
68
|
+
redis = (redisFn ?? makeRedisClient)(url);
|
|
69
|
+
redis.on?.("error", () => {});
|
|
70
|
+
}
|
|
71
|
+
try {
|
|
72
|
+
const { readLiveHosts } = await import("./host-registry.mjs");
|
|
73
|
+
// Both reads at once, so a Valkey that does not answer costs one timeout, not two.
|
|
74
|
+
// `prune: false`: this command writes nothing, not even the registry reader's tidying of a dead member. `now` is the
|
|
75
|
+
// report's clock: a row's age decides how far its running jobs count (`LIVE_FRESH_MS`).
|
|
76
|
+
const [fleet, read] = await Promise.all([
|
|
77
|
+
redis ? (readLiveHostsFn ?? readLiveHosts)(redis, { now, timeoutMs: FLEET_READ_TIMEOUT_MS, prune: false }).catch((error) => ({ unreachable: error?.message ?? "registry unreadable" })) : { unreachable: null },
|
|
78
|
+
readCapacityRecords({ redis, logsDir, sinceMs: windowStartMs, nowMs, retentionDays, localHost, noMirrorReason: refusedNote, timeoutMs: FLEET_READ_TIMEOUT_MS, ...(fs ? { fs } : {}) }),
|
|
79
|
+
]);
|
|
80
|
+
if (named && (fleet.unreachable || read.mirrorState.startsWith("unreachable"))) return fail(`could not read Valkey at ${urlShown(url)}: ${fleet.unreachable ? `host registry ${fleet.unreachable}` : `run mirror ${read.mirrorState}`}`);
|
|
81
|
+
const coverage = { ...read.coverage, reason: [read.coverage.reason, fleet.unreachable ? `host registry unreadable (${fleet.unreachable})` : null].filter(Boolean).join("; ") || null };
|
|
82
|
+
const report = computeCapacity({ records: read.records, live: fleet.hosts ?? [], windowStartMs, nowMs, bucketMs: window.bucketMs, coverage });
|
|
83
|
+
const shown = host === undefined ? { report } : onlyHost(report, host);
|
|
84
|
+
if (shown.unknown) return fail(`no host named ${JSON.stringify(host)} in the last ${since}${shown.unknown.length > 0 ? ` (hosts: ${shown.unknown.join(", ")})` : ""}`);
|
|
85
|
+
write(json ? `${JSON.stringify(shown.report)}\n` : capacityText(shown.report, { since }));
|
|
86
|
+
return 0;
|
|
87
|
+
} finally {
|
|
88
|
+
redis?.disconnect?.();
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* The report cut to one host (`--host`, and the admin's `dispatch_capacity`): `{ report }`, or `{ unknown: [names] }` when
|
|
94
|
+
* the report has no such host. The coverage then says what the shown host's history is: its start, its cut and its
|
|
95
|
+
* counts. What cannot be put on a host (a record with no readable host, the live rows' running jobs, the reasons a
|
|
96
|
+
* source was not read) stays.
|
|
97
|
+
*/
|
|
98
|
+
export function onlyHost(report, host) {
|
|
99
|
+
const known = report.hosts.map((h) => h.name);
|
|
100
|
+
if (!known.includes(host)) return { unknown: known };
|
|
101
|
+
const shown = report.hosts.find((h) => h.name === host);
|
|
102
|
+
const { fromMs, source: _source, truncated, ...counts } = shown.coverage;
|
|
103
|
+
return { report: { ...report, hosts: [shown], coverage: { ...report.coverage, fromMs, truncated, ...counts, historyNotShared: report.coverage.historyNotShared.filter((n) => n === host) } } };
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* What this command reads of the deployment: the logs directory (config.mjs `logsDirPath`, the worker's rule), the
|
|
108
|
+
* retention (30 days unless set; 0 keeps the files) and this host's name (`PI_WORKER_NAME`, else the hostname), or
|
|
109
|
+
* `{ problem }` for a retention the worker would refuse to boot with.
|
|
110
|
+
*/
|
|
111
|
+
export async function deploymentFacts(env) {
|
|
112
|
+
const raw = env.PI_LOG_RETENTION_DAYS;
|
|
113
|
+
if (raw !== undefined && raw !== "" && !/^\d{1,6}$/.test(raw)) return { problem: `PI_LOG_RETENTION_DAYS must be a non-negative integer (got ${JSON.stringify(raw)})` };
|
|
114
|
+
const { logsDirPath, defaultWorkerName, WORKER_NAME_RE } = await import("./config.mjs");
|
|
115
|
+
const named = env.PI_WORKER_NAME;
|
|
116
|
+
return { logsDir: logsDirPath(env), retentionDays: raw === undefined || raw === "" ? 30 : Number(raw), localHost: typeof named === "string" && WORKER_NAME_RE.test(named) ? named : defaultWorkerName() };
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** A span in the largest whole units that say it: `40s`, `6m`, `2h 5m`, `3d 4h`. */
|
|
120
|
+
export function durationText(ms) {
|
|
121
|
+
const s = Math.round(ms / 1000);
|
|
122
|
+
if (s < 60) return `${s}s`;
|
|
123
|
+
const m = Math.floor(s / 60);
|
|
124
|
+
if (m < 60) return `${m}m`;
|
|
125
|
+
const h = Math.floor(m / 60);
|
|
126
|
+
if (h < 24) return m % 60 === 0 ? `${h}h` : `${h}h ${m % 60}m`;
|
|
127
|
+
const d = Math.floor(h / 24);
|
|
128
|
+
return h % 24 === 0 ? `${d}d` : `${d}d ${h % 24}h`;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/** Thousandths as a percentage, at most one decimal: 125 is `12.5%`, 1000 is `100%`. */
|
|
132
|
+
export function percentText(perMille) {
|
|
133
|
+
return `${Math.floor(perMille / 10)}${perMille % 10 === 0 ? "" : `.${perMille % 10}`}%`;
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** Thousandths as a number with at most one decimal, rounded half up: 2149 is `2.1`, 2000 is `2`. */
|
|
137
|
+
export function milliText(milli) {
|
|
138
|
+
const tenths = Math.floor((milli + 50) / 100);
|
|
139
|
+
return `${Math.floor(tenths / 10)}${tenths % 10 === 0 ? "" : `.${tenths % 10}`}`;
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
/** Part of whole as a per-mille, rounded half up; 0 when there is no whole. The panel's HOSTS view reads its shares by it too. */
|
|
143
|
+
export const share = (part, whole) => (whole > 0 ? Math.floor((part * 2000 + whole) / (whole * 2)) : 0);
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Part of whole as a percentage, the way every surface prints a share of time: `percentText(share(...))`, except that a
|
|
147
|
+
* share that is there but rounds to nothing reads `under 0.1%` and one short of the whole that rounds to all of it reads
|
|
148
|
+
* `over 99.9%`. "full 0%" beside "peak 3 of 3" said the host was never full when it was, briefly.
|
|
149
|
+
*/
|
|
150
|
+
export function shareText(part, whole) {
|
|
151
|
+
const p = share(part, whole);
|
|
152
|
+
if (p === 0 && part > 0 && whole > 0) return "under 0.1%";
|
|
153
|
+
if (p === 1000 && part < whole) return "over 99.9%";
|
|
154
|
+
return percentText(p);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* C0 and C1 control characters, which a terminal acts on (cursor moves, a title, a cleared screen). The report admits
|
|
159
|
+
* only worker names and project ids, which hold none; this strips them anyway before anything reaches the terminal, so
|
|
160
|
+
* a reader that ever admits more cannot hand a record's author the operator's screen.
|
|
161
|
+
*/
|
|
162
|
+
const CONTROL = /[\u0000-\u0009\u000b-\u001f\u007f-\u009f]/g;
|
|
163
|
+
|
|
164
|
+
/** The human report: one block per host, then what the fleet's history covers. Control characters stripped. */
|
|
165
|
+
export function capacityText(report, { since }) {
|
|
166
|
+
const lines = [];
|
|
167
|
+
for (const h of report.hosts) lines.push(...hostLines(h, since, report.coverage));
|
|
168
|
+
if (report.hosts.length === 0) lines.push(`${capitalized(noRunText(report.coverage, `in the last ${since}`))}.`);
|
|
169
|
+
lines.push(...coverageLines(report.coverage));
|
|
170
|
+
return `${lines.join("\n").replace(CONTROL, "")}\n`;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
const plural = (n, word) => `${n} ${word}${n === 1 ? "" : "s"}`;
|
|
174
|
+
const capitalized = (text) => `${text.slice(0, 1).toUpperCase()}${text.slice(1)}`;
|
|
175
|
+
|
|
176
|
+
/**
|
|
177
|
+
* What a report with no host says, `when` being "in the last 24h" or "in this window": that no host ran a job only when
|
|
178
|
+
* the whole window was read: no reason (no source missing, no record skipped), not this host's files alone, the run
|
|
179
|
+
* mirror's history not cut (`truncated`, which for a report with no host is the mirror's own), and no record left
|
|
180
|
+
* uncounted (unreadable, or without a host). Otherwise only that the history read here holds no run, since a host whose
|
|
181
|
+
* runs were not read may have run many. The CLI, the panel's HOSTS view and the insights page say it in these words.
|
|
182
|
+
*/
|
|
183
|
+
export function noRunText(cov, when) {
|
|
184
|
+
const whole = (cov?.reason ?? null) === null && cov?.source !== "local" && cov?.truncated !== true && !(cov?.unreadable > 0) && !(cov?.withoutHost > 0);
|
|
185
|
+
return whole ? `no host ran a job ${when}` : `no run in the history read here ${when}`;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Why a host's history starts inside the window when the run mirror cut it (`coverage.truncated`), in the words the CLI
|
|
190
|
+
* uses; the insights page restates them (admin/src/insights-html.mjs `capGapWhy`, pinned by its test): the mirror
|
|
191
|
+
* holds nothing older, because it started then (`runs:since`: a new deployment, or a Valkey that lost its keys), or its
|
|
192
|
+
* cap or a peer's shorter retention cut it.
|
|
193
|
+
*/
|
|
194
|
+
export const MIRROR_CUT_WHY = "the run mirror holds nothing older: it started then, or its cap or a peer's shorter retention cut it";
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* Why a host's history is not here (`hosts[].notShared`), in one sentence for the CLI, the tool and doctor, so none of
|
|
198
|
+
* them blames the name for a Valkey that did not answer: `unread` names the reason the run mirror was not read (the
|
|
199
|
+
* coverage's `reason`); only `unnamed` says it is a worker without `PI_WORKER_NAME`.
|
|
200
|
+
*/
|
|
201
|
+
export function notSharedWhy(h, coverage) {
|
|
202
|
+
if (h?.notShared === "unread") return `the run mirror was not read${coverage?.reason ? ` (${coverage.reason})` : ""}, so its runs are not here`;
|
|
203
|
+
return "no source here holds its runs (a worker without PI_WORKER_NAME writes no run mirror)";
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function hostLines(h, since, coverage) {
|
|
207
|
+
const lines = [`Host ${h.name}, last ${since}`];
|
|
208
|
+
if (h.coveredMs === 0) {
|
|
209
|
+
lines.push(` no history here${h.shared ? "" : `: ${notSharedWhy(h, coverage)}`}`);
|
|
210
|
+
return lines;
|
|
211
|
+
}
|
|
212
|
+
const f = hostFacts(h);
|
|
213
|
+
const missing = f.missing !== null ? ` (${f.missing} of the window has no history)` : "";
|
|
214
|
+
lines.push(` busy ${f.busy}, idle ${f.idle}${missing}`);
|
|
215
|
+
const of = f.slots !== null ? ` of ${f.slots}` : "";
|
|
216
|
+
const basis = f.basis === null ? "" : ` (${f.basis})`;
|
|
217
|
+
const full = f.full !== null ? `, full ${f.full} of the time` : "";
|
|
218
|
+
lines.push(` slots: avg ${f.avg}${of}, peak ${f.peak}${of}${full}${basis}`);
|
|
219
|
+
lines.push(` promised: ${f.memory}, ${f.cpu}`);
|
|
220
|
+
if (f.cpuUsed !== null) lines.push(` CPU used: ${f.cpuUsed}`);
|
|
221
|
+
lines.push(f.wait !== null ? ` wait for a slot: ${f.wait}` : " wait for a slot: no run recorded one");
|
|
222
|
+
if (f.projects.length > 0) lines.push(` projects by run time: ${f.projects.join(", ")}`);
|
|
223
|
+
for (const caveat of hostCaveats(h.coverage)) lines.push(` ${caveat}`);
|
|
224
|
+
lines.push(` ${historyNotes(h).join("; ")}`);
|
|
225
|
+
return lines;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/**
|
|
229
|
+
* A host's headline numbers in words, as the CLI prints them (`hostLines`) and the insights page shows them, so the two
|
|
230
|
+
* cannot say a number differently: busy and idle of the covered time, the share of the window with no history (null
|
|
231
|
+
* when none), slots on average and at peak with the slot count (null when unknown), time full (null when not known),
|
|
232
|
+
* the basis note, what memory and CPU were promised against, CPU used (null when not measured), the wait (null when no
|
|
233
|
+
* run recorded one), and the projects by run time with the rest summed. For a host with covered time only.
|
|
234
|
+
*/
|
|
235
|
+
export function hostFacts(h) {
|
|
236
|
+
const c = h.capacity;
|
|
237
|
+
const memOf = Number.isSafeInteger(c.memMiB) && c.memMiB > 0 ? ` of the ${formatMemory(c.memMiB)} budget` : "";
|
|
238
|
+
const hostCpus = c.cpus !== null ? `the host's ${c.cpus} CPUs` : "the host's CPUs";
|
|
239
|
+
const promiseOf = Number.isSafeInteger(c.cpuCenti) && c.cpuCenti > 0 ? `the ${formatCpus(c.cpuCenti)} CPU budget` : `${hostCpus} (no CPU budget)`;
|
|
240
|
+
const projects = h.projects.map((p) => `${p.project ?? "(no project)"} ${durationText(p.runMs)}`);
|
|
241
|
+
if (h.projects.length > 0 && h.otherProjects) projects.push(`${plural(h.otherProjects.count, "other")} ${durationText(h.otherProjects.runMs)}`);
|
|
242
|
+
return {
|
|
243
|
+
...busyIdleText(h.busyMs, h.coveredMs),
|
|
244
|
+
missing: h.missingMs > 0 ? shareText(h.missingMs, h.missingMs + h.coveredMs) : null,
|
|
245
|
+
avg: milliText(h.avgMilli ?? 0),
|
|
246
|
+
slots: c.slots,
|
|
247
|
+
peak: h.peak,
|
|
248
|
+
full: h.fullMs !== null ? shareText(h.fullMs, h.coveredMs) : null,
|
|
249
|
+
basis: basisNote(c),
|
|
250
|
+
memory: h.promisedMemPerMille !== null ? `memory ${percentText(h.promisedMemPerMille)}${memOf}` : `memory: no budget${c.memMiB === "off" ? " (off)" : ""}`,
|
|
251
|
+
cpu: h.promisedCpuPerMille !== null ? `CPU ${percentText(h.promisedCpuPerMille)} of ${promiseOf}` : "CPU: no budget or CPU count known",
|
|
252
|
+
cpuUsed: h.usedCpuPerMille !== null ? `${percentText(h.usedCpuPerMille)} of ${hostCpus}` : null,
|
|
253
|
+
wait: h.waits.n > 0 ? `p50 ${durationText(h.waits.p50Ms)}, p95 ${durationText(h.waits.p95Ms)} (${plural(h.waits.n, "run")})` : null,
|
|
254
|
+
projects,
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* How a host's slot count was judged, as the CLI words it after "slots: avg ... of N", or null when the count is the one
|
|
260
|
+
* every run recorded: the panel's HOSTS view qualifies its "of N" and its full share with the same words.
|
|
261
|
+
*/
|
|
262
|
+
export function basisNote(c) {
|
|
263
|
+
if (c?.basis === "recorded") return c.changed ? "the capacity changed in the window: each part is judged by the one in force then, the newest is shown" : null;
|
|
264
|
+
return c?.basis === "current" ? "current setting, no run recorded one" : "slot count unknown";
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* What a host's numbers leave out or infer, one sentence each, from its coverage counts (`hosts[].coverage`): refusals
|
|
269
|
+
* before a slot, retries and stalls that under-count busy time, the jobs running now and those not counted, an
|
|
270
|
+
* unreadable job list, orphans. The CLI prints one per line; the panel's HOSTS view prints the same sentences.
|
|
271
|
+
*/
|
|
272
|
+
export function hostCaveats(cov) {
|
|
273
|
+
const out = [];
|
|
274
|
+
if (cov.refusedBeforeSlot > 0) out.push(`${plural(cov.refusedBeforeSlot, "job")} refused before a slot`);
|
|
275
|
+
// Two sentences, each true of THIS host (phase 4's review): an earlier attempt is counted on the host that ran it, which
|
|
276
|
+
// need not be the retry's, so "N retried runs: M earlier attempts counted" read 0 on the retry's host while the
|
|
277
|
+
// attempt was counted on another.
|
|
278
|
+
if (cov.retried > 0) out.push(`${plural(cov.retried, "retried run")}: ${cov.retried === 1 ? "its earlier attempts are" : "their earlier attempts are"} counted on the host that ran them, where that host's history is here and covers them; an attempt whose record was not kept is not counted, so busy time can be under-counted`);
|
|
279
|
+
if (cov.earlier > 0) out.push(cov.earlier === 1 ? "1 earlier attempt of a retried run counted here, from the record its retry kept" : `${cov.earlier} earlier attempts of retried runs counted here, from the records their retries kept`);
|
|
280
|
+
if (cov.live > 0) out.push(`${plural(cov.live, "job")} running now, counted as busy up to now (or the host's last beat)`);
|
|
281
|
+
if (cov.liveNotCounted > 0) out.push(`${cov.liveNotCounted} more running now ${cov.liveNotCounted === 1 ? "is" : "are"} not counted (not listed by its row, or its history is not shared), so busy time can be under-counted`);
|
|
282
|
+
if (cov.liveUnreadable > 0) out.push("its list of running jobs could not be read, so none of them is counted and how many run is unknown");
|
|
283
|
+
if (cov.orphans > 0) out.push(`${plural(cov.orphans, "orphaned container")} (a stop that did not take) still held by the budget, counted by ${cov.orphans === 1 ? "its record" : "their records"} up to the stop`);
|
|
284
|
+
if (cov.stalledRepick > 0) out.push(`${cov.stalledRepick} ${cov.stalledRepick === 1 ? "run was" : "runs were"} picked up again after a stall: the first pickup's time is not counted`);
|
|
285
|
+
return out;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** Where a host's history comes from (`hosts[].coverage.source`), in words. */
|
|
289
|
+
export function historySourceText(source) {
|
|
290
|
+
return source === "local" ? "this host's files" : source === "mirror" ? "the run mirror" : source === "mirror+local" ? "the run mirror and this host's files" : "the run records";
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
/**
|
|
294
|
+
* Where a host's history comes from and what of it is missing or inferred, one clause each: its source, the start of a
|
|
295
|
+
* history that begins inside the window (earlier time is neither busy nor idle), and the records counted by inference
|
|
296
|
+
* or left out of promised or CPU used. The CLI joins them with "; ", as does the panel's HOSTS view.
|
|
297
|
+
*/
|
|
298
|
+
export function historyNotes(h) {
|
|
299
|
+
const cov = h.coverage;
|
|
300
|
+
const notes = [`history from ${historySourceText(cov.source)}`];
|
|
301
|
+
if (h.missingMs > 0) notes.push(`from ${new Date(cov.fromMs).toISOString().slice(0, 16).replace("T", " ")} UTC on${cov.truncated ? ` (${MIRROR_CUT_WHY})` : ""}, earlier time counted as neither busy nor idle`);
|
|
302
|
+
const legacy = cov.legacyOccupied + cov.legacyRefused;
|
|
303
|
+
if (legacy > 0) notes.push(`${plural(legacy, "record")} from before capacity was recorded, inferred (${cov.legacyOccupied} held a slot, ${cov.legacyRefused} refused)`);
|
|
304
|
+
if (cov.withoutSize > 0) notes.push(`${cov.withoutSize} without a size (not in promised)`);
|
|
305
|
+
if (cov.withoutResources > 0) notes.push(`${cov.withoutResources} without a CPU measurement (not in CPU used)`);
|
|
306
|
+
if (cov.capacityOutOfRange > 0) notes.push(`${plural(cov.capacityOutOfRange, "record")} giving a capacity no host can have, that value read as unknown`);
|
|
307
|
+
if (cov.cpuClamped > 0) notes.push(`${cov.cpuClamped} reporting more CPU than the job could use, read at that most`);
|
|
308
|
+
return notes;
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/**
|
|
312
|
+
* The fleet's records that no host's numbers hold, one clause each: unreadable records, records without a host, carried
|
|
313
|
+
* earlier attempts not counted. The CLI's coverage line and the panel's HOSTS view say them in these words.
|
|
314
|
+
*/
|
|
315
|
+
export function fleetRecordNotes(cov) {
|
|
316
|
+
const notes = [];
|
|
317
|
+
if (cov.unreadable > 0) notes.push(`${plural(cov.unreadable, "record")} unreadable, not counted`);
|
|
318
|
+
if (cov.withoutHost > 0) notes.push(`${cov.withoutHost} without a host, not counted`);
|
|
319
|
+
if (cov.earlierDropped > 0) notes.push(`${plural(cov.earlierDropped, "carried earlier attempt")} not counted (not valid, beyond the 4 a record keeps, or overlapping its own run)`);
|
|
320
|
+
return notes;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* What the fleet's history covers, one clause each: where it comes from, the hosts whose history is not here, the records
|
|
325
|
+
* no host's numbers hold, the running jobs not counted or unknown, and why a source was not read. The CLI joins them into
|
|
326
|
+
* its coverage line (`coverageLines`); the insights page shows each on its own, so a long host list can never push the
|
|
327
|
+
* others out. `namesShown` cuts the host list to that many names and says how many more (the CLI names them all).
|
|
328
|
+
*/
|
|
329
|
+
export function coverageNotes(cov, { namesShown = Infinity } = {}) {
|
|
330
|
+
const notes = [`history: ${cov.source === "local" ? "this host's files only" : cov.source === "mirror" ? "the run mirror" : cov.source === "mirror+local" ? "the run mirror and this host's files" : "the records given"}`];
|
|
331
|
+
const names = cov.historyNotShared;
|
|
332
|
+
if (names.length > namesShown) notes.push(`${plural(names.length, "host")} whose history is not here: ${names.slice(0, namesShown).join(", ")} and ${names.length - namesShown} more`);
|
|
333
|
+
else if (names.length > 0) notes.push(`not shared here: ${names.join(", ")}`);
|
|
334
|
+
notes.push(...fleetRecordNotes(cov));
|
|
335
|
+
// No row that lists anything (the registry not read, or every row gone until its next beat), or none for the reading
|
|
336
|
+
// host while its runs are here: those running jobs are not counted, and the line says so rather than leaving a busy
|
|
337
|
+
// host reading as one that runs nothing. The other rows' count stands.
|
|
338
|
+
const noRows = cov.liveUnreadable === 0 && cov.running === null;
|
|
339
|
+
if (cov.liveUnreadable > 0) notes.push(`the running jobs of ${plural(cov.liveUnreadable, "host")} unreadable, not counted, so how many run now is unknown`);
|
|
340
|
+
else if (noRows) notes.push("no live row lists the jobs running now, so they are not known");
|
|
341
|
+
if (cov.running !== null && cov.running > 0) notes.push(`${plural(cov.running, "job")} running now${cov.liveNotCounted > 0 ? `, ${cov.liveNotCounted} of them not counted until ${cov.liveNotCounted === 1 ? "it ends" : "they end"}` : ", counted up to now"}`);
|
|
342
|
+
if (typeof cov.liveRowMissing === "string" && !noRows) notes.push(`no live row read for this host (${cov.liveRowMissing}), so the jobs it runs now are not known`);
|
|
343
|
+
if (cov.reason) notes.push(cov.reason);
|
|
344
|
+
return notes;
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Busy and idle as printed, `{ busy, idle }`: busy is `shareText(busy, covered)`, and idle is its complement, so the two
|
|
349
|
+
* printed numbers always sum to 100% (each rounded on its own, 123.5 and 876.5 per mille printed 12.4% and 87.7%).
|
|
350
|
+
* Covered time is busy or idle and nothing else (missing time is neither), so the complement is the idle share.
|
|
351
|
+
*/
|
|
352
|
+
export function busyIdleText(busyMs, coveredMs) {
|
|
353
|
+
const busy = shareText(busyMs, coveredMs);
|
|
354
|
+
if (!(coveredMs > 0)) return { busy, idle: busy };
|
|
355
|
+
if (busyMs <= 0) return { busy, idle: "100%" };
|
|
356
|
+
if (busyMs >= coveredMs) return { busy, idle: "0%" };
|
|
357
|
+
const p = share(busyMs, coveredMs);
|
|
358
|
+
if (p === 0) return { busy, idle: "over 99.9%" };
|
|
359
|
+
if (p === 1000) return { busy, idle: "under 0.1%" };
|
|
360
|
+
return { busy, idle: percentText(1000 - p) };
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/** What no surface of the report can see (REQ-CAPACITY-INSIGHTS): every one says it in these words. */
|
|
364
|
+
export const JOBS_ONLY = "Jobs only: a machine busy with other work reads as idle.";
|
|
365
|
+
|
|
366
|
+
/** The CLI's last two lines: what the fleet's history covers, and what this report cannot see. */
|
|
367
|
+
export function coverageLines(cov) {
|
|
368
|
+
const notes = coverageNotes(cov);
|
|
369
|
+
// The insights page draws this clause as its own warning, so it is added here, not in `coverageNotes`.
|
|
370
|
+
if (cov.truncated === true) notes.splice(1, 0, TRUNCATED_NOTE);
|
|
371
|
+
return [`Coverage: ${notes.join("; ")}.`, JOBS_ONLY];
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/** The fleet's history cut by the run mirror (`coverage.truncated`), as the CLI's coverage line says it. */
|
|
375
|
+
export const TRUNCATED_NOTE = "history truncated: the run mirror holds nothing older for at least one host, so its earlier time is counted as neither busy nor idle";
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The run records a capacity report reads (issue #599, DES-CAPACITY-FROM-RECORDS), and what they cover. Two sources,
|
|
3
|
+
* merged by the one rule the panel already reads the fleet's runs by (run-mirror.mjs `mergeRuns`):
|
|
4
|
+
*
|
|
5
|
+
* - THE RUN MIRROR, where workers declared names (`runs:index`, `runs:rec:*`, `runs:horizon` and `runs:since`,
|
|
6
|
+
* run-mirror.mjs): every named host's records. Read in a fixed number of bounded round trips: ONE script
|
|
7
|
+
* (`READ_SCRIPT`: the size, the oldest score, the horizon, the mirror's start and the members over the window, one
|
|
8
|
+
* snapshot, so a flush between them cannot mix two states), then MGET in chunks of `CAPACITY_MGET_CHUNK`. A body
|
|
9
|
+
* over 256 KiB is skipped and counted, as a local file is. READ-ONLY: unlike `readMirroredRuns`, this reader prunes nothing, so a
|
|
10
|
+
* report can never change what another surface shows.
|
|
11
|
+
* - THE LOCAL FILES in the logs directory, size-records.mjs' pattern: an mtime prefilter (a file last written before
|
|
12
|
+
* the window, less a day of skew, is not opened) and the 256 KiB cap per file (`SIZING_RECORD_MAX_BYTES`), a larger
|
|
13
|
+
* file skipped and counted.
|
|
14
|
+
*
|
|
15
|
+
* WHAT EACH COVERS, never more than it can show, so a report never reads lost history as idle time. The two sources
|
|
16
|
+
* cover different hosts, so each states its own start and `computeCapacity` judges every host by the sources that hold
|
|
17
|
+
* all of its runs:
|
|
18
|
+
* - the local files hold this host's runs (and every host's, on a shared logs directory) back to
|
|
19
|
+
* `PI_LOG_RETENTION_DAYS` (0 keeps them). An absent or unreadable directory covers nothing (`local: null`): a host
|
|
20
|
+
* whose files are gone has no history here, not an idle one;
|
|
21
|
+
* - the mirror holds the named hosts' runs back to the LATEST of: the mirror's own start (`runs:since`, trusted only
|
|
22
|
+
* while the run that vouches for it is in the index; without a trusted one, the index's oldest run, since an index
|
|
23
|
+
* written before that key, recreated by a worker that does not know it, or whose key was lost, vouches for nothing
|
|
24
|
+
* older), the deepest window any writer keeps
|
|
25
|
+
* (`mirrorWindowMs(0)`), the fleet horizon (`runs:horizon`, raised by every writer whose trim removed runs, by its
|
|
26
|
+
* OWN retention or the count cap, so a peer with a short retention cuts everyone's history and says so), the oldest
|
|
27
|
+
* run at the `RUNS_INDEX_MAX` cap when that run ended after the window began, and the newest run whose body has
|
|
28
|
+
* expired while its index member stayed (its writer has not trimmed since). Any of these past the window's start
|
|
29
|
+
* makes it `truncated`. The start is the one bound that survives losing the mirror's keys: a flushed or replaced
|
|
30
|
+
* Valkey, or an index that expired, loses the horizon with the runs, and the next write recreates an index whose
|
|
31
|
+
* other bounds all reach back past the loss, so without it a peer read every lost run as idle.
|
|
32
|
+
*
|
|
33
|
+
* An absent index is `off` (no named worker, or none since the index expired), and an unreachable or slow Valkey is
|
|
34
|
+
* read as no mirror, both with the reason stated: the local files are still read, and the report says it is local
|
|
35
|
+
* only. Every Valkey call is bounded (`bounded`, host-registry.mjs' reason: this client QUEUES a command while offline
|
|
36
|
+
* rather than rejecting it, so an unbounded await is a hang, not an error).
|
|
37
|
+
*
|
|
38
|
+
* Never throws.
|
|
39
|
+
*/
|
|
40
|
+
|
|
41
|
+
import { readdirSync, readFileSync, statSync } from "node:fs";
|
|
42
|
+
import { join } from "node:path";
|
|
43
|
+
import { READ_SCRIPT, RUNS_HORIZON, RUNS_HORIZON_MEMBER, RUNS_INDEX, RUNS_INDEX_MAX, RUNS_SINCE, mergeRuns, mirrorWindowMs, runRecordKey } from "./run-mirror.mjs";
|
|
44
|
+
import { SIZING_RECORD_MAX_BYTES } from "./size-records.mjs";
|
|
45
|
+
|
|
46
|
+
/** How many record bodies one MGET asks for: a full mirror is ten round trips, never one 5,000-key command. */
|
|
47
|
+
export const CAPACITY_MGET_CHUNK = 500;
|
|
48
|
+
/** How long one Valkey call may take before the mirror is read as unreachable. */
|
|
49
|
+
export const CAPACITY_OP_TIMEOUT_MS = 2_000;
|
|
50
|
+
|
|
51
|
+
const DAY_MS = 24 * 60 * 60 * 1000;
|
|
52
|
+
|
|
53
|
+
/** Reject rather than wait forever (host-registry.mjs `bounded`). */
|
|
54
|
+
function bounded(promise, ms) {
|
|
55
|
+
return new Promise((resolve, reject) => {
|
|
56
|
+
const t = setTimeout(() => reject(new Error("timeout")), ms);
|
|
57
|
+
Promise.resolve(promise).then(
|
|
58
|
+
(v) => (clearTimeout(t), resolve(v)),
|
|
59
|
+
(e) => (clearTimeout(t), reject(e)),
|
|
60
|
+
);
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* The mirror's records that ended after `sinceMs`: `{ records, state, fromMs, truncated, skipped }`, `state` "ok", "off"
|
|
66
|
+
* or "unreachable (<reason>)", `fromMs` where its history starts (see the header). Never throws.
|
|
67
|
+
*/
|
|
68
|
+
export async function readMirrorWindow(redis, { sinceMs, nowMs, timeoutMs = CAPACITY_OP_TIMEOUT_MS, chunk = CAPACITY_MGET_CHUNK } = {}) {
|
|
69
|
+
const none = (state) => ({ records: [], state, fromMs: null, truncated: false, skipped: 0 });
|
|
70
|
+
if (!redis) return none("off");
|
|
71
|
+
try {
|
|
72
|
+
const snap = await bounded(redis.eval(READ_SCRIPT, 3, RUNS_INDEX, RUNS_HORIZON, RUNS_SINCE, RUNS_HORIZON_MEMBER, sinceMs), timeoutMs);
|
|
73
|
+
if (!Array.isArray(snap)) throw new Error("the index did not answer a list");
|
|
74
|
+
const size = Number(snap[0]);
|
|
75
|
+
if (!Number.isSafeInteger(size) || size <= 0) return none("off");
|
|
76
|
+
const num = (v) => (typeof v === "string" && /^\d{1,16}$/.test(v) ? Number(v) : typeof v === "number" && Number.isSafeInteger(v) ? v : NaN);
|
|
77
|
+
const oldestMs = num(snap[1]);
|
|
78
|
+
const horizonMs = num(snap[2]);
|
|
79
|
+
// The start counts only while the run that vouches for it is in the index (`RUNS_SINCE`).
|
|
80
|
+
const startMs = snap[4] === 1 ? num(snap[3]) : NaN;
|
|
81
|
+
const ids = snap[5];
|
|
82
|
+
if (!Array.isArray(ids) || ids.length % 2 !== 0) throw new Error("the index did not answer a list");
|
|
83
|
+
const members = [];
|
|
84
|
+
for (let i = 0; i < ids.length; i += 2) members.push({ id: ids[i], score: Number(ids[i + 1]) });
|
|
85
|
+
const records = [];
|
|
86
|
+
let skipped = 0;
|
|
87
|
+
let expiredMs = NaN;
|
|
88
|
+
for (let i = 0; i < members.length; i += chunk) {
|
|
89
|
+
const part = members.slice(i, i + chunk);
|
|
90
|
+
const bodies = await bounded(redis.mget(...part.map((m) => runRecordKey(m.id))), timeoutMs);
|
|
91
|
+
part.forEach((m, k) => {
|
|
92
|
+
const raw = Array.isArray(bodies) ? bodies[k] : null;
|
|
93
|
+
if (typeof raw !== "string" || raw === "") {
|
|
94
|
+
// Expired while its member stayed: a run this index once held and no longer shows, so the history of the
|
|
95
|
+
// mirror cannot start before it (the members are newest first, so the first one found is the newest).
|
|
96
|
+
if (!Number.isFinite(expiredMs) && Number.isFinite(m.score)) expiredMs = m.score;
|
|
97
|
+
return;
|
|
98
|
+
}
|
|
99
|
+
if (Buffer.byteLength(raw, "utf8") > SIZING_RECORD_MAX_BYTES) {
|
|
100
|
+
skipped++;
|
|
101
|
+
return;
|
|
102
|
+
}
|
|
103
|
+
try {
|
|
104
|
+
const record = JSON.parse(raw);
|
|
105
|
+
if (record !== null && typeof record === "object" && !Array.isArray(record)) records.push(record);
|
|
106
|
+
} catch {
|
|
107
|
+
// unparseable is not showable
|
|
108
|
+
}
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
let fromMs = Math.max(sinceMs, nowMs - mirrorWindowMs(0));
|
|
112
|
+
// The mirror's own start; without a trusted one, the oldest run the index holds (`RUNS_SINCE`): never the window's
|
|
113
|
+
// start.
|
|
114
|
+
if (Number.isFinite(startMs)) fromMs = Math.max(fromMs, startMs);
|
|
115
|
+
else if (Number.isFinite(oldestMs)) fromMs = Math.max(fromMs, oldestMs);
|
|
116
|
+
if (Number.isFinite(horizonMs)) fromMs = Math.max(fromMs, horizonMs);
|
|
117
|
+
if (size >= RUNS_INDEX_MAX && Number.isFinite(oldestMs)) fromMs = Math.max(fromMs, oldestMs);
|
|
118
|
+
if (Number.isFinite(expiredMs)) fromMs = Math.max(fromMs, expiredMs);
|
|
119
|
+
fromMs = Math.min(fromMs, nowMs);
|
|
120
|
+
return { records, state: "ok", fromMs, truncated: fromMs > sinceMs, skipped };
|
|
121
|
+
} catch (err) {
|
|
122
|
+
return none(`unreachable (${err?.message ?? "error"})`);
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* This host's records that ended after `sinceMs`: `{ records, skipped, unreachable, absent }`. `fs` is `{ readdirSync,
|
|
128
|
+
* statSync, readFileSync }`.
|
|
129
|
+
*/
|
|
130
|
+
export function readLocalWindow(logsDir, { sinceMs, fs = { readdirSync, readFileSync, statSync } } = {}) {
|
|
131
|
+
let names;
|
|
132
|
+
try {
|
|
133
|
+
names = fs.readdirSync(logsDir);
|
|
134
|
+
} catch (err) {
|
|
135
|
+
const absent = err?.code === "ENOENT";
|
|
136
|
+
return { records: [], skipped: 0, unreachable: absent ? null : `logs dir unreadable (${err?.code ?? "read-error"})`, absent };
|
|
137
|
+
}
|
|
138
|
+
const records = [];
|
|
139
|
+
let skipped = 0;
|
|
140
|
+
for (const name of names) {
|
|
141
|
+
if (typeof name !== "string" || !name.endsWith(".json")) continue;
|
|
142
|
+
try {
|
|
143
|
+
const path = join(logsDir, name);
|
|
144
|
+
const st = fs.statSync(path);
|
|
145
|
+
// A record file is written when its run ends, so one last written before the window (less a day of skew) holds
|
|
146
|
+
// no run that reaches into it.
|
|
147
|
+
if (!st.isFile() || st.mtimeMs < sinceMs - DAY_MS) continue;
|
|
148
|
+
if (st.size > SIZING_RECORD_MAX_BYTES) {
|
|
149
|
+
skipped++;
|
|
150
|
+
continue;
|
|
151
|
+
}
|
|
152
|
+
const buf = fs.readFileSync(path);
|
|
153
|
+
if (buf.length > SIZING_RECORD_MAX_BYTES) {
|
|
154
|
+
skipped++; // it grew between the stat and the read
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
const record = JSON.parse(buf.toString("utf8"));
|
|
158
|
+
if (record === null || typeof record !== "object" || Array.isArray(record) || typeof record.jobId !== "string") continue;
|
|
159
|
+
const end = Date.parse(record.endedAt ?? "");
|
|
160
|
+
if (Number.isFinite(end) && end > sinceMs) records.push(record);
|
|
161
|
+
} catch {
|
|
162
|
+
// unparseable, or reaped between the listing and the read
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
return { records, skipped, unreachable: null, absent: false };
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* The records of the window `[sinceMs, nowMs]` from both sources, merged, and what each covers: `{ records, mirrorState,
|
|
170
|
+
* coverage }`,
|
|
171
|
+
* `coverage` being `{ source, reason, localHost, localHosts, local: { fromMs } | null, mirror: { fromMs, truncated,
|
|
172
|
+
* hosts } | null, skipped }` as `computeCapacity` reads it. `redis` may be null (no Valkey to ask); `noMirrorReason` then
|
|
173
|
+
* says why, in place of "no run mirror". `localHost` is this host's name. Never throws.
|
|
174
|
+
*/
|
|
175
|
+
export async function readCapacityRecords({ redis = null, logsDir, sinceMs, nowMs, fs, retentionDays = 0, localHost = null, noMirrorReason = null, timeoutMs = CAPACITY_OP_TIMEOUT_MS, chunk = CAPACITY_MGET_CHUNK } = {}) {
|
|
176
|
+
const mirror = await readMirrorWindow(redis, { sinceMs, nowMs, timeoutMs, chunk });
|
|
177
|
+
const local = readLocalWindow(logsDir, { sinceMs, ...(fs ? { fs } : {}) });
|
|
178
|
+
const mirrored = mirror.state === "ok";
|
|
179
|
+
const days = Number(retentionDays);
|
|
180
|
+
const localFromMs = Math.min(nowMs, Number.isFinite(days) && days > 0 ? Math.max(sinceMs, nowMs - days * DAY_MS) : sinceMs);
|
|
181
|
+
const localRead = local.unreachable === null && !local.absent;
|
|
182
|
+
const reasons = [];
|
|
183
|
+
if (!mirrored) {
|
|
184
|
+
if (!redis && noMirrorReason) reasons.push(noMirrorReason);
|
|
185
|
+
else if (mirror.state === "off") reasons.push("no run mirror: only this host's files were read");
|
|
186
|
+
else reasons.push(`run mirror ${mirror.state}: only this host's files were read`);
|
|
187
|
+
}
|
|
188
|
+
if (local.absent) reasons.push("no logs directory here, so this host's own runs are not read");
|
|
189
|
+
if (local.unreachable) reasons.push(local.unreachable);
|
|
190
|
+
const skipped = local.skipped + mirror.skipped;
|
|
191
|
+
if (skipped > 0) reasons.push(`${skipped} record${skipped === 1 ? "" : "s"} over ${SIZING_RECORD_MAX_BYTES / 1024} KiB skipped`);
|
|
192
|
+
const hostsOf = (records) => [...new Set(records.map((r) => r.host).filter((h) => typeof h === "string" && h !== ""))].sort();
|
|
193
|
+
return {
|
|
194
|
+
// One per job id, the later end winning (a retry can land on another host); `Infinity` keeps every run, since the
|
|
195
|
+
// window, not a page, bounds this read.
|
|
196
|
+
records: mergeRuns(local.records, mirror.records, { limit: Infinity }),
|
|
197
|
+
// "ok", "off" or "unreachable (<reason>)": a caller that named this Valkey fails on the last.
|
|
198
|
+
mirrorState: mirror.state,
|
|
199
|
+
coverage: {
|
|
200
|
+
source: mirrored ? (localRead ? "mirror+local" : "mirror") : "local",
|
|
201
|
+
reason: reasons.length > 0 ? reasons.join("; ") : null,
|
|
202
|
+
localHost,
|
|
203
|
+
localHosts: hostsOf(local.records),
|
|
204
|
+
local: localRead ? { fromMs: localFromMs } : null,
|
|
205
|
+
mirror: mirrored ? { fromMs: mirror.fromMs, truncated: mirror.truncated, hosts: hostsOf(mirror.records) } : null,
|
|
206
|
+
skipped,
|
|
207
|
+
},
|
|
208
|
+
};
|
|
209
|
+
}
|