@edgehero/pi-dispatch 4.0.0 → 4.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -39,6 +39,158 @@ import { UNREADABLE_RECORD } from "./run-history.mjs";
39
39
  /** The index: sanitized jobId -> the run's end (or start) in millis. */
40
40
  export const RUNS_INDEX = "runs:index";
41
41
 
42
+ /**
43
+ * The fleet's HISTORY HORIZON (issue #599): a one-member ZSET (`trim`) whose score is the newest instant before which a
44
+ * writer has trimmed runs out of the index. Every writer trims the SHARED index by its OWN retention (and the shared
45
+ * count cap), so a host with a short `PI_LOG_RETENTION_DAYS` removes every host's older runs; a reader cannot know
46
+ * that from its own settings, and without this key it read the gap as idle time. Raised only, never lowered (`ZADD
47
+ * GT`), so concurrent writers agree on the latest cut without a lock.
48
+ */
49
+ export const RUNS_HORIZON = "runs:horizon";
50
+ /** The horizon's one member. */
51
+ export const RUNS_HORIZON_MEMBER = "trim";
52
+
53
+ /**
54
+ * WHEN THIS MIRROR STARTED (issue #599): a hash, `at` the instant in millis from which the index holds every run that
55
+ * ended, and `member` a run of the index that vouches for it. Nothing above says it: the reader's other bounds (the
56
+ * depth, the horizon, the cap, an expired body) all live in keys that a flushed or replaced Valkey, or an index that
57
+ * expired, loses together with the runs, and the next write then recreates an index that claimed the whole window, so a
58
+ * peer read every lost run as idle.
59
+ *
60
+ * A start is TRUSTED only while its `member` is in the index. The start can outlive its index (a worker of 4.0.1 sets
61
+ * the index's expiry to its own retention while the start keeps the deepest window; a pruning reader can remove every
62
+ * member), and an older worker can then recreate the index without knowing this key: a member that is still there is
63
+ * the one proof that the index the start describes is the index that exists. Rejected: also trusting it while
64
+ * `runs:horizon` is at or past it, since the horizon outlives a lost index just the same.
65
+ * - The write that finds no index starts it NOW (the write's time), vouched for by the run it adds.
66
+ * - A write that finds an index whose start it cannot trust (none, or its member is gone) starts it at the index's
67
+ * oldest run before this write, vouched for by that run: the most the index can say, so time before it reads as
68
+ * missing. A trusted start is kept.
69
+ * - Whatever removes the vouching run while the start is trusted (a trim, `PRUNE_SCRIPT`) hands it to the oldest run
70
+ * that remains, `at` unchanged, and deletes the start with the last run. So no trim moves `at`, earlier or later.
71
+ * It expires with the index (one `PEXPIRE` each, in the same script). A reader that finds no trusted start starts the
72
+ * mirror at the index's oldest run.
73
+ */
74
+ export const RUNS_SINCE = "runs:since";
75
+
76
+ /**
77
+ * Lua shared by the three scripts (KEYS[1] the index, KEYS[3] the start): whether the start is a hash, whether its member
78
+ * is held, and handing a trusted start on once its member is removed. A start that is not a hash (a string from a
79
+ * development build, or a hand SET) is read as absent, and the writing scripts delete it first (`sinceClean`), so a
80
+ * stray value can never make HGET fail with WRONGTYPE and stop the mirror.
81
+ */
82
+ const SINCE_LUA = `
83
+ local function sinceIsHash()
84
+ local t = redis.call("TYPE", KEYS[3])
85
+ return (t.ok or t) == "hash"
86
+ end
87
+ local function sinceClean()
88
+ if redis.call("EXISTS", KEYS[3]) == 1 and not sinceIsHash() then redis.call("DEL", KEYS[3]) end
89
+ end
90
+ local function sinceHeld()
91
+ if not sinceIsHash() then return false end
92
+ local member = redis.call("HGET", KEYS[3], "member")
93
+ return member and redis.call("ZSCORE", KEYS[1], member) and true or false
94
+ end
95
+ local function sinceHandOn()
96
+ local first = redis.call("ZRANGE", KEYS[1], 0, 0)
97
+ if first[1] then redis.call("HSET", KEYS[3], "member", first[1]) else redis.call("DEL", KEYS[3]) end
98
+ end
99
+ `;
100
+
101
+ /**
102
+ * The writer's index write and trim, in one script so the member, the start (`RUNS_SINCE`) and the horizon it raises
103
+ * are atomic (issue #599). KEYS: the index, the horizon and the start; ARGV: the age cutoff in millis, the count cap,
104
+ * the expiry all three keys get, the horizon's member, the run's score, its id, and the write's time in millis.
105
+ *
106
+ * - The start is checked and set as `RUNS_SINCE` says, from the index as it was before this write, and the member is
107
+ * added (`ZADD`).
108
+ * - By age (scores below the cutoff, exclusive) and by count (all but the newest `cap`). A trim that REMOVED something
109
+ * raises the horizon with `ZADD GT`: to the cutoff, or after a count trim to the oldest score that remains, which is
110
+ * conservative (a removed run ended at or before it). Nothing removed, nothing written: a fleet whose writers trim
111
+ * nothing keeps no horizon. A trim that removed the start's member hands the start on.
112
+ * - The keys expire after the DEEPEST window any reader asks for (`mirrorWindowMs(0)`), never the writer's own. The
113
+ * index's members are trimmed by score anyway; a short-retention writer that set the shared index's TTL to its own
114
+ * window made the whole index expire on a quiet day, every peer's history with it and no horizon to say so. The
115
+ * expiry still rolls with traffic: a fleet that runs nothing for that long has nothing to show.
116
+ */
117
+ export const TRIM_SCRIPT = `${SINCE_LUA}
118
+ sinceClean()
119
+ local oldest = redis.call("ZRANGE", KEYS[1], 0, 0, "WITHSCORES")
120
+ if not oldest[2] then
121
+ redis.call("DEL", KEYS[3])
122
+ redis.call("HSET", KEYS[3], "at", ARGV[7], "member", ARGV[6])
123
+ elseif not (redis.call("HGET", KEYS[3], "at") and sinceHeld()) then
124
+ redis.call("DEL", KEYS[3])
125
+ redis.call("HSET", KEYS[3], "at", string.format("%d", tonumber(oldest[2])), "member", oldest[1])
126
+ end
127
+ redis.call("ZADD", KEYS[1], ARGV[5], ARGV[6])
128
+ redis.call("PEXPIRE", KEYS[3], ARGV[3])
129
+ local removed = redis.call("ZREMRANGEBYSCORE", KEYS[1], "-inf", "(" .. ARGV[1])
130
+ local horizon = nil
131
+ if removed > 0 then horizon = tonumber(ARGV[1]) end
132
+ if redis.call("ZREMRANGEBYRANK", KEYS[1], 0, -tonumber(ARGV[2]) - 1) > 0 then
133
+ local kept = redis.call("ZRANGE", KEYS[1], 0, 0, "WITHSCORES")
134
+ if kept[2] then
135
+ local score = tonumber(kept[2])
136
+ if horizon == nil or score > horizon then horizon = score end
137
+ end
138
+ end
139
+ if horizon then
140
+ redis.call("ZADD", KEYS[2], "GT", string.format("%d", horizon), ARGV[4])
141
+ redis.call("PEXPIRE", KEYS[2], ARGV[3])
142
+ if not sinceHeld() then sinceHandOn() end
143
+ end
144
+ redis.call("PEXPIRE", KEYS[1], ARGV[3])
145
+ return horizon and string.format("%d", horizon) or false
146
+ `;
147
+
148
+ /**
149
+ * A pruning reader's removal of members whose bodies are gone (`readMirroredRuns`), in one script so the cut is
150
+ * recorded with it (issue #599). The run a removed member stood for is no longer shown, so the time up to it is no
151
+ * longer whole: the horizon is raised (`ZADD GT`) to the highest score removed, as `TRIM_SCRIPT` raises it for a trim,
152
+ * and a trusted start whose member is removed is handed on. Without it, the reader's "expired body" bound was erased by
153
+ * the very read that saw it, and the time before it read as idle. KEYS: the index, the horizon and the start; ARGV: the
154
+ * horizon's expiry, its member, then the ids. Returns the highest score removed, or false.
155
+ */
156
+ export const PRUNE_SCRIPT = `${SINCE_LUA}
157
+ sinceClean()
158
+ -- Handed on only when the start was trusted BEFORE this removal: a start whose run was already gone (an older worker
159
+ -- recreated the index under it) must not be revalidated by handing it to a run of the new index.
160
+ local held = sinceHeld()
161
+ local top = nil
162
+ for i = 3, #ARGV do
163
+ local score = redis.call("ZSCORE", KEYS[1], ARGV[i])
164
+ if score then
165
+ redis.call("ZREM", KEYS[1], ARGV[i])
166
+ score = tonumber(score)
167
+ if top == nil or score > top then top = score end
168
+ end
169
+ end
170
+ if top then
171
+ redis.call("ZADD", KEYS[2], "GT", string.format("%d", top), ARGV[2])
172
+ redis.call("PEXPIRE", KEYS[2], ARGV[1])
173
+ if held and not sinceHeld() then sinceHandOn() end
174
+ end
175
+ return top and string.format("%d", top) or false
176
+ `;
177
+
178
+ /**
179
+ * The capacity reader's one snapshot of the index (issue #599): its size, its oldest score, the horizon, the start's
180
+ * `at` and whether its member is held, and the members that ended after ARGV[2] (exclusive) with their scores, newest
181
+ * first. One script, so a flush or a write between these reads cannot mix two states into one report. KEYS: the index,
182
+ * the horizon and the start; ARGV: the horizon's member and the window's start.
183
+ */
184
+ export const READ_SCRIPT = `${SINCE_LUA}
185
+ local size = redis.call("ZCARD", KEYS[1])
186
+ if size == 0 then return {0} end
187
+ local oldest = redis.call("ZRANGE", KEYS[1], 0, 0, "WITHSCORES")
188
+ local horizon = redis.call("ZSCORE", KEYS[2], ARGV[1])
189
+ local at = sinceIsHash() and redis.call("HGET", KEYS[3], "at") or false
190
+ local range = redis.call("ZREVRANGEBYSCORE", KEYS[1], "+inf", "(" .. ARGV[2], "WITHSCORES")
191
+ return {size, oldest[2] or false, horizon, at, sinceHeld() and 1 or 0, range}
192
+ `;
193
+
42
194
  /** One run's own bytes. */
43
195
  export const runRecordKey = (sanitizedJobId) => `runs:rec:${sanitizedJobId}`;
44
196
 
@@ -106,17 +258,13 @@ export function makeRunMirror({ redis, retentionDays, now = () => Date.now(), lo
106
258
  const score = Number.isFinite(at) ? at : now();
107
259
  const body = JSON.stringify(record);
108
260
  await bounded(redis.set(runRecordKey(sanitizedJobId), body, "PX", windowMs), timeoutMs);
109
- await bounded(redis.zadd(RUNS_INDEX, score, sanitizedJobId), timeoutMs);
110
- // Trimmed by the WRITER, twice: by age, and by count. Two `ZREMRANGE`s against a run that
111
- // took minutes is free, and it means no reader has to pay for a backlog it did not create.
112
- await bounded(redis.zremrangebyscore(RUNS_INDEX, "-inf", `(${now() - windowMs}`), timeoutMs);
113
- await bounded(redis.zremrangebyrank(RUNS_INDEX, 0, -indexMax - 1), timeoutMs);
114
- // ROLLING expiry, deliberately unlike `budget.mjs`'s set-once rule and deliberately like
115
- // `pi-dispatch:sched-stalls:<schedulerId>`. A budget window must not be pushed forward by traffic or a busy
116
- // day never resets; an ACTIVITY index should roll with traffic, because that is what it
117
- // describes. A fleet that stops running jobs loses its index one window later, which is
118
- // correct: there is nothing left to show.
119
- await bounded(redis.pexpire(RUNS_INDEX, windowMs), timeoutMs);
261
+ // Indexed, and trimmed by the WRITER, twice: by age, and by count, with the mirror's start (`RUNS_SINCE`) and
262
+ // the fleet's horizon written by the same script (`TRIM_SCRIPT`), so no crash or timeout can land between a
263
+ // member and the record of where the index starts, or between a cut and the record of it. A script, not a
264
+ // MULTI, because both depend on what the index held: the start on whether it existed, the horizon on what a
265
+ // trim REMOVED and what remains.
266
+ const writtenMs = now();
267
+ await bounded(redis.eval(TRIM_SCRIPT, 3, RUNS_INDEX, RUNS_HORIZON, RUNS_SINCE, writtenMs - windowMs, indexMax, mirrorWindowMs(0), RUNS_HORIZON_MEMBER, score, sanitizedJobId, writtenMs), timeoutMs);
120
268
  warned = false;
121
269
  return true;
122
270
  } catch (err) {
@@ -180,7 +328,8 @@ export async function readMirroredRuns(redis, { limit = 50, sinceMs = 0, now = (
180
328
  }
181
329
  if (stale.length > 0) {
182
330
  try {
183
- await bounded(redis.zrem(RUNS_INDEX, ...stale), timeoutMs);
331
+ // Removed with the cut recorded (`PRUNE_SCRIPT`): the time up to a pruned run is no longer whole.
332
+ await bounded(redis.eval(PRUNE_SCRIPT, 3, RUNS_INDEX, RUNS_HORIZON, RUNS_SINCE, mirrorWindowMs(0), RUNS_HORIZON_MEMBER, ...stale), timeoutMs);
184
333
  } catch {
185
334
  // best-effort: a straggler in the index costs one skipped row next time, never a wrong one
186
335
  }
@@ -42,7 +42,8 @@
42
42
  * A size is rounded UP to a step: 256m up to 2g, 512m up to 8g, then 1g.
43
43
  * CPUs: fewer than `SUGGEST_MIN_SAMPLES` runs is not enough runs; the p95 cores used (CPU time over wall time) below
44
44
  * 0.4x the current cpus lowers to 1.25x that p95, never below 1.25x the window's largest cores used, nor 0.25;
45
- * otherwise it fits. A FACT (no call) rides along when the median run was throttled more than 25% of its wall time.
45
+ * otherwise it fits. A FACT (no call) rides along when the median run was throttled more than 25% of its wall time,
46
+ * read over the runs that report throttled time and only when at least `SUGGEST_MIN_SAMPLES` of them do.
46
47
  * The wall time is the record's pickup-to-end span, which includes the clone, so cores used read slightly LOW.
47
48
  *
48
49
  * THE CAP. A raise never goes past `cap` (this host's budget per dimension, or where that is off or unknown the host's
@@ -119,9 +120,10 @@ const isCount = (v) => Number.isSafeInteger(v) && v >= 0;
119
120
  /**
120
121
  * One record as a suggestion reads it, or null when it is not evidence: `{ at, size, memPeak, oom, wallUsec, cpuUsec,
121
122
  * throttledUsec, memFullUsec }`. `memPeak` (bytes, clamped to the record's own limit) is null when absent or not a
122
- * count; `wallUsec` is null unless the span is readable; `cpuUsec` and `throttledUsec` are null unless both can be read
123
- * with a wall time (CPU time clamped to the CPU ceiling over the wall, throttled time to the wall); `memFullUsec` is
124
- * null unless it can be read with a wall time.
123
+ * count; `wallUsec` is null unless the span is readable; `cpuUsec` is null unless it can be read with a wall time (CPU
124
+ * time clamped to the CPU ceiling over the wall), and `throttledUsec` likewise on its own (clamped to the wall): cgroup v2
125
+ * writes `throttled_usec` in `cpu.stat` only where the cpu controller is enabled, so a run without it is still a CPU
126
+ * sample, and only the throttle fact leaves it out. `memFullUsec` is null unless it can be read with a wall time.
125
127
  */
126
128
  function evidenceOf(record, project, nowMs) {
127
129
  if (record === null || typeof record !== "object" || Array.isArray(record)) return null;
@@ -136,11 +138,10 @@ function evidenceOf(record, project, nowMs) {
136
138
  const start = typeof record.startedAt === "string" ? Date.parse(record.startedAt) : NaN;
137
139
  const wallMs = Number.isFinite(start) ? at - start : NaN;
138
140
  const wallUsec = Number.isSafeInteger(wallMs) && wallMs > 0 && wallMs <= SUGGEST_MAX_WALL_MS ? wallMs * 1000 : null;
139
- let cpu = { cpuUsec: null, throttledUsec: null };
140
- if (wallUsec !== null && isCount(r.cpuUsec) && isCount(r.throttledUsec)) {
141
- const cpuCap = (wallUsec * JOB_CPUS_CEILING_CENTI) / 100;
142
- cpu = { cpuUsec: Math.min(r.cpuUsec, cpuCap), throttledUsec: Math.min(r.throttledUsec, wallUsec) };
143
- }
141
+ const cpu = {
142
+ cpuUsec: wallUsec !== null && isCount(r.cpuUsec) ? Math.min(r.cpuUsec, (wallUsec * JOB_CPUS_CEILING_CENTI) / 100) : null,
143
+ throttledUsec: wallUsec !== null && isCount(r.cpuUsec) && isCount(r.throttledUsec) ? Math.min(r.throttledUsec, wallUsec) : null,
144
+ };
144
145
  // unclamped: it is only ever compared (exactly, in BigInt) with 1% of the wall, and a stall past the wall is above it
145
146
  const memFullUsec = wallUsec !== null && isCount(r.memFullUsec) ? r.memFullUsec : null;
146
147
  if (memPeak === null && cpu.cpuUsec === null) return null;
@@ -174,7 +175,8 @@ function suggestMemory(runs, current, cap) {
174
175
  const largestOom = oomSizes.length > 0 ? Math.max(...oomSizes) : null;
175
176
  // at the limit (90% of it or more: peak x 10 >= limit x 9) AND stalled for memory more than 1% of the wall.
176
177
  const pressured = relevant.filter((e) => !productAbove(e.size.memMiB * MIB, 9, e.memPeak, 10) && e.memFullUsec !== null && productAbove(e.memFullUsec, 100, e.wallUsec, 1)).length;
177
- const evidence = { samples, p95MiB: p95 === null ? null : Math.ceil(p95 / MIB), maxPeakMiB: maxPeak === null ? null : Math.ceil(maxPeak / MIB), ooms, largestOomMiB: largestOom, pressured };
178
+ // `smaller`: the measured runs below the current size, which a chart draws and this verdict does not count.
179
+ const evidence = { samples, smaller: measured.length - samples, p95MiB: p95 === null ? null : Math.ceil(p95 / MIB), maxPeakMiB: maxPeak === null ? null : Math.ceil(maxPeak / MIB), ooms, largestOomMiB: largestOom, pressured };
178
180
  const fact = pressured > 0 ? "pressure" : null;
179
181
  const out = (reason, suggested, more = {}) => ({ current, suggested: suggested === null || suggested === current ? null : suggested, reason, evidence, fact, held: null, wanted: null, ...more });
180
182
  if (ooms > 0) {
@@ -198,12 +200,17 @@ function suggestCpus(runs, current) {
198
200
  const samples = relevant.length;
199
201
  // ordered by the exact fraction, never a float: a/b < c/d <=> a x d < c x b.
200
202
  const byFraction = (num) => (x, y) => (productAbove(num(y), x.wallUsec, num(x), y.wallUsec) ? -1 : productAbove(num(x), y.wallUsec, num(y), x.wallUsec) ? 1 : 0);
201
- const throttle = samples > 0 ? rank([...relevant].sort(byFraction((e) => e.throttledUsec)), 50) : null;
203
+ // The throttle fact reads only the runs whose cgroup reported throttled time (see `evidenceOf`), and needs as many of
204
+ // them as a suggestion needs runs: one throttled run among twenty is not "the median run".
205
+ const throttled = relevant.filter((e) => e.throttledUsec !== null);
206
+ const throttle = throttled.length >= SUGGEST_MIN_SAMPLES ? rank([...throttled].sort(byFraction((e) => e.throttledUsec)), 50) : null;
202
207
  const busy = samples > 0 ? rank([...relevant].sort(byFraction((e) => e.cpuUsec)), 95) : null;
203
208
  const busiest = measured.length > 0 ? [...measured].sort(byFraction((e) => e.cpuUsec)).at(-1) : null;
204
209
  const coresOf = (e) => (e === null ? null : Math.ceil((e.cpuUsec * 100) / e.wallUsec));
205
210
  const evidence = {
206
211
  samples,
212
+ smaller: measured.length - samples,
213
+ throttledSamples: throttled.length,
207
214
  p95CoresCenti: coresOf(busy),
208
215
  maxCoresCenti: coresOf(busiest),
209
216
  throttledPct: throttle === null ? null : Math.round((throttle.throttledUsec * 100) / throttle.wallUsec),
@@ -211,7 +218,7 @@ function suggestCpus(runs, current) {
211
218
  const out = (reason, suggested, fact = null) => ({ current, suggested: suggested === null || suggested === current ? null : suggested, reason, evidence, fact, held: null, wanted: null });
212
219
  if (samples < SUGGEST_MIN_SAMPLES) return out("not-enough-runs", null);
213
220
  // throttled more than 25% of the wall time (throttled x 4 > wall): a fact about the host's ceiling, never a call.
214
- const fact = productAbove(throttle.throttledUsec, 4, throttle.wallUsec, 1) ? "ceiling" : null;
221
+ const fact = throttle !== null && productAbove(throttle.throttledUsec, 4, throttle.wallUsec, 1) ? "ceiling" : null;
215
222
  // p95 cores below 0.4 x cpus: cpuUsec / wall < 0.4 x current / 100, that is cpuUsec x 1000 < 4 x current x wall.
216
223
  if (productAbove(4 * current, busy.wallUsec, busy.cpuUsec, 1000)) {
217
224
  const lower = Math.max(roundCpusUp(quarterMoreCenti(busy)), roundCpusUp(quarterMoreCenti(busiest)));
@@ -399,6 +406,19 @@ export const coresText = (centi) => `${formatCpus(centi)} core${centi === 100 ?
399
406
  /** CPUs in words: `1 CPU`, `0.5 CPUs`, `2 CPUs`. */
400
407
  export const cpusText = (centi) => `${formatCpus(centi)} CPU${centi === 100 ? "" : "s"}`;
401
408
 
409
+ /**
410
+ * A "not enough runs" verdict in words for the panel's PROJECTS view and the dispatch_limit_edit preview (the insights
411
+ * page restates it; doctor says the same counts through `suggestionEvidence`): how many runs at the current size or
412
+ * larger it counted, and how many smaller ones it did not, which for memory the chart beside it draws:
413
+ * `not enough runs at 4g (0; 20 at smaller sizes)`. `sizeText` formats the dimension's size.
414
+ */
415
+ export function notEnoughRunsText(dim, sizeText) {
416
+ const e = dim?.evidence ?? {};
417
+ const samples = Number.isSafeInteger(e.samples) ? e.samples : 0;
418
+ const smaller = Number.isSafeInteger(e.smaller) && e.smaller > 0 ? `; ${e.smaller} at smaller sizes` : "";
419
+ return `not enough runs at ${sizeText(dim.current)} (${samples}${smaller})`;
420
+ }
421
+
402
422
  /**
403
423
  * The evidence of a suggestion in words, for doctor, the panel and the insights page alike: `{ memory, cpu, memoryHeld,
404
424
  * memoryFact, cpuFact }`, each a short clause (empty when it does not apply), such as `2 runs ended oom-killed (the
@@ -415,12 +435,12 @@ export function suggestionEvidence(suggestion) {
415
435
  const cores = `p95 ${coresText(ce.p95CoresCenti ?? 0)} used, largest ${coresText(ce.maxCoresCenti ?? 0)}, over ${plural(ce.samples, "run")}`;
416
436
  const memWords = {
417
437
  "oom-killed": `${plural(me.ooms, "run")} ended oom-killed (the largest size killed ${formatMemory(me.largestOomMiB ?? 0)})`,
418
- "not-enough-runs": `${me.samples} of the ${SUGGEST_MIN_SAMPLES} runs with measurements it needs`,
438
+ "not-enough-runs": `${me.samples} of the ${SUGGEST_MIN_SAMPLES} runs with measurements at ${formatMemory(m.current ?? 0)} or larger it needs${me.smaller > 0 ? `, ${me.smaller} more at smaller sizes` : ""}`,
419
439
  oversized: peaks,
420
440
  fits: peaks,
421
441
  };
422
442
  const cpuWords = {
423
- "not-enough-runs": `${ce.samples} of the ${SUGGEST_MIN_SAMPLES} runs with measurements it needs`,
443
+ "not-enough-runs": `${ce.samples} of the ${SUGGEST_MIN_SAMPLES} runs with measurements at ${cpusText(c.current ?? 0)} or more it needs${ce.smaller > 0 ? `, ${ce.smaller} more at smaller sizes` : ""}`,
424
444
  underused: cores,
425
445
  fits: cores,
426
446
  };
package/src/start.mjs CHANGED
@@ -25,6 +25,7 @@ import { builtinModel, checkModelsKnown } from "./model-catalog.mjs";
25
25
  import { capabilityTokens, serializeCaps } from "./capabilities.mjs";
26
26
  import { cronFingerprint } from "./fingerprint.mjs";
27
27
  import { makeHostRegistry } from "./host-registry.mjs";
28
+ import { liveJobsFields, liveJobsOf } from "./live-jobs.mjs";
28
29
  import { budgetField, readUserServiceLimits } from "./host-budget.mjs";
29
30
  import { makeCpuReserve, reservePlan } from "./cpu-reserve.mjs";
30
31
  import { makeImagePreflight } from "./image-preflight.mjs";
@@ -63,7 +64,7 @@ import { PODMAN_RESTART_HOLD_EXPIRED, makePodmanServiceReader, onceFs, makeRootf
63
64
  import { makeRunContainer } from "./run-container.mjs";
64
65
  import { resolveProviderCredential } from "./env-allowlist.mjs";
65
66
  import { makeSecretsResolver } from "./secrets.mjs";
66
- import { buildRecord, EXIT_OOM_KILLED, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, RUNNER_POLICY_REASONS, sanitizeJobId } from "./run-history.mjs";
67
+ import { buildRecord, EXIT_OOM_KILLED, makeFindPreviousRun, makeLogReaper, makeLogSink, makeReadRecord, makeRecordWriter, makeSettledRecord, newerRecord, RUNNER_POLICY_REASONS, sanitizeJobId, UNREADABLE_RECORD } from "./run-history.mjs";
67
68
  import { makeRunMirror, readMirroredRecord } from "./run-mirror.mjs";
68
69
  import { readOverlay, resolveSettings } from "./runtime-settings.mjs";
69
70
  import { usdFingerprint } from "./dollar-fingerprint.mjs";
@@ -1368,7 +1369,7 @@ export async function startWorker(
1368
1369
  // would be bytes nothing reads. That is also what keeps a single-host deployment byte-identical, since
1369
1370
  // no job then issues a single extra Valkey command.
1370
1371
  const runMirror = config.workerNameDeclared ? makeRunMirrorFn({ redis, retentionDays: config.logRetentionDays, log }) : null;
1371
- const recordRun = ({ job, result, error, startedAt, endedAt, project, size = null }) => {
1372
+ const recordRun = ({ job, result, error, startedAt, endedAt, project, size = null, capacity = null, earlier = null }) => {
1372
1373
  // The project (issue #499) was resolved at the pickup gate and rides here as `project` (an id or null), so a live
1373
1374
  // edit of projects.json mid-run cannot make the record disagree with what the job was counted against. A record
1374
1375
  // path that ends BEFORE the pickup gate (the wait gate's refusals) passes none, and resolves from the live ref
@@ -1378,7 +1379,7 @@ export async function startWorker(
1378
1379
  // four `recordRun` call sites byte-unchanged and `buildRecord` a pure function of its arguments.
1379
1380
  // The default venue rides the same way and for the same reason (#277): it is the value the registry
1380
1381
  // below is built with, so the record resolves a job's venue exactly as dispatch does.
1381
- const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId, size });
1382
+ const record = buildRecord({ job, result, error, startedAt, endedAt, host: config.workerName, defaultBackend: config.defaultBackend, project: projectId, size, capacity, earlier });
1382
1383
  writeRecord(record);
1383
1384
  // STRICTLY AFTER the file, and deliberately not awaited. After, because a crash between the two must
1384
1385
  // leave a record with no fleet row rather than a fleet row with no record -- the mirror is a VIEW,
@@ -1399,10 +1400,23 @@ export async function startWorker(
1399
1400
  // The record read back, for a job the queue lost the lock of after it finished (DES-TERMINAL-COMMENTS-AND-FAILURE-HOOK,
1400
1401
  // CONST-RETRY-INFRA-ONLY): this host's own file first, then the fleet's copy where a mirror is armed, because the
1401
1402
  // host that meets the stalled job need not be the one that ran it. A single host has no mirror and needs none.
1403
+ const readRecord = makeReadRecord({ logsDir: config.logsDir });
1402
1404
  const settledRecord = makeSettledRecord({
1403
- readRecord: makeReadRecord({ logsDir: config.logsDir }),
1405
+ readRecord,
1404
1406
  readMirrored: runMirror ? (jobId) => readMirroredRecordFn(redis, sanitizeJobId(jobId)) : null,
1405
1407
  });
1408
+ // The record a retry (or a pickup after a stall) is about to replace (issue #599), read BOTH ways: this host's file and
1409
+ // the fleet's copy where a mirror is armed. Both, because each host keeps its own file: a job that ran on A, then B,
1410
+ // then A again finds A's stale first attempt locally while B's second is in the mirror. `newerRecord` keeps the higher
1411
+ // attempt (a tie: the later end). Its slot interval rides the new record as `earlier`. Never rejects: unreadable or
1412
+ // unreachable is no record from that side.
1413
+ const previousRecord = async (jobId) => {
1414
+ const read = readRecord(jobId);
1415
+ const local = read === UNREADABLE_RECORD ? null : read;
1416
+ if (!runMirror) return local;
1417
+ const mirrored = await readMirroredRecordFn(redis, sanitizeJobId(jobId)).catch(() => null);
1418
+ return newerRecord(local, mirrored === UNREADABLE_RECORD ? null : mirrored);
1419
+ };
1406
1420
 
1407
1421
  // INT-CONFIG-OVERLAY-CONTRACT: the worker reads the runtime-settings overlay at EACH job start, so this
1408
1422
  // closure -- not a value frozen at boot -- is what the processor calls per job. It resolves the fourteen
@@ -1618,6 +1632,18 @@ export async function startWorker(
1618
1632
  const snap = worker?.hostBudget?.snapshot?.();
1619
1633
  return Number.isSafeInteger(snap?.[key]) ? String(snap[key]) : "";
1620
1634
  };
1635
+ // Issue #599, phase 2: the jobs this host runs now and its budget's orphans, oldest first, as the row's two fields
1636
+ // (`live-jobs.mjs`). Without a budget there is no ledger, so no orphan, and the list is the in-flight map alone.
1637
+ const liveJobsNow = () => {
1638
+ const read = (fn) => {
1639
+ try {
1640
+ return fn() ?? [];
1641
+ } catch {
1642
+ return [];
1643
+ }
1644
+ };
1645
+ return liveJobsFields(liveJobsOf({ running: read(() => worker?.runningJobs?.list?.()), budgetEntries: read(() => worker?.hostBudget?.entries?.()) }));
1646
+ };
1621
1647
  // NOT awaited, and that is load-bearing rather than an optimisation. `makeRedisClient` sets
1622
1648
  // `maxRetriesPerRequest: null` -- required for BullMQ's blocking connections -- which means a command
1623
1649
  // issued against an unreachable server QUEUES FOREVER instead of rejecting. Awaiting the first beat
@@ -1684,6 +1710,14 @@ export async function startWorker(
1684
1710
  heldMemMiB: () => snapshotField("heldMemMiB"),
1685
1711
  heldCpuCenti: () => snapshotField("heldCpuCenti"),
1686
1712
  budgetRunning: () => snapshotField("running"),
1713
+ // Issue #599, phase 2: the jobs waiting for the budget ("" without one), and the jobs this host runs now: `jobs` is
1714
+ // JSON of at most 32 `{ id, p, m, c, at, o? }` (job id, project id or null, size integers, admission millis, 1 for an
1715
+ // orphan), oldest first, `jobsMore` how many are not listed. Ids and integers only (live-jobs.mjs says why that
1716
+ // meets the content rule). Each thunk builds the list again; the beat reads every thunk with no await between, so
1717
+ // the two agree. "" before the worker exists.
1718
+ waiters: () => snapshotField("waiters"),
1719
+ jobs: () => (worker ? liveJobsNow().jobs : ""),
1720
+ jobsMore: () => (worker ? liveJobsNow().jobsMore : ""),
1687
1721
  budgetHolds: () => snapshotField("holds"),
1688
1722
  budgetOrphans: () => snapshotField("orphans"),
1689
1723
  // whether the boot listing of the job containers left from before this worker started has been read
@@ -2059,6 +2093,7 @@ export async function startWorker(
2059
2093
  redis,
2060
2094
  recordRun,
2061
2095
  settledRecord,
2096
+ previousRecord,
2062
2097
  extraClosers,
2063
2098
  // REQ-SCOPED-PAUSE-WINDOWS: the processor defers a job whose folder/repo is inside an active window.
2064
2099
  // Reads the live-reloaded ref, so an operator edit takes effect on the next job without a restart.
package/src/wait-for.mjs CHANGED
@@ -40,6 +40,13 @@ function configError(message) {
40
40
  */
41
41
  export const WAIT_CONDITION_KEYS = ["after", "profile"];
42
42
 
43
+ /**
44
+ * Every reason the wait gate refuses a job with (index.mjs `refuseWait`), each before the job holds a slot. A test
45
+ * holds this list equal to the processor's calls; the capacity report reads it to tell such a refusal from a run on
46
+ * a record from before `capacity` was recorded (issue #599).
47
+ */
48
+ export const WAIT_REFUSAL_REASONS = Object.freeze(["wait-superseded", "wait-unreadable", "wait-profile-unknown", "wait-after-beyond-max", "wait-expired", "wait-refused", "wait-unanswerable"]);
49
+
43
50
  /**
44
51
  * The ceiling on conditions per trigger. Four, on `SECRETS_MAX`'s reasoning rather than `REPLICAS_MAX`':
45
52
  * this multiplies no spend, but every `profile` condition is a subprocess evaluated before the container,
@@ -0,0 +1,25 @@
1
+ /**
2
+ * The worker name rule (issue #57), in a module with NO imports, so a pure reader (the capacity report, issue #599) can
3
+ * hold a record's `host` to it without loading config.mjs and its fs and os. `config.mjs` re-exports the name.
4
+ */
5
+
6
+ /**
7
+ * What a worker may call itself (issue #57). The CHARACTER CLASS is `sanitizeJobId`'s
8
+ * (`[A-Za-z0-9._-]`), reused rather than invented so this project has one name-safe alphabet -- but that
9
+ * function is a REPLACER, not a validator, so the three rules around the class are NEW and are claimed
10
+ * as new here rather than borrowed:
11
+ *
12
+ * - a leading alphanumeric, which is what refuses `..` and a leading `-` that reads as a flag;
13
+ * - a 64-character ceiling, because the name is a Valkey key segment and a log field on every line;
14
+ * - no `.json`/`.log` tail, which is not decoration. The class contains the dot, so `prod.json` is
15
+ * otherwise a legal name -- and a later slice writes a per-host marker file into `PI_LOGS_DIR`,
16
+ * where `<something>.json` is parsed as a run record by the admin and DELETED by the log reaper.
17
+ * A name is refused here rather than escaped there, because the escape would have to be remembered
18
+ * at every site that ever composes a filename from this value.
19
+ *
20
+ * The class is `:`-free, `,`-free and `#`-free, which is what lets the name be a Valkey key segment
21
+ * UNHASHED. That is the point of validating instead of hashing (`scopeKeyPrefix` does the opposite for
22
+ * a folder path, which was never chosen for key-safety and cannot be refused): the whole value of a host
23
+ * registry is that `HGETALL host:h:mac-mini-1` is readable by a human.
24
+ */
25
+ export const WORKER_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/;