faberun 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -280,22 +280,35 @@ export function clearTierExhaustion(state) {
280
280
  }
281
281
 
282
282
  /**
283
+ * Settle what a closed invocation produced, and start whatever the outcome
284
+ * earns next.
285
+ *
286
+ * The two maps are two different things, and conflating them is what let a run
287
+ * exceed its own `maxParallel` (measured 2026-09-21, run
288
+ * state-location-and-routing-economics-13): `closed` is the job this call owns
289
+ * -- the scheduler hands one node's job at a time so two settlements never
290
+ * interleave -- while `running` is the run's live dispatch authority, the very
291
+ * map the scheduler's `maxParallel - running.size` counts. Anything started
292
+ * here lands in `running` and is accounted from that instant; nothing started
293
+ * here may be settled by this call.
294
+ *
283
295
  * @param {ValidatedContract} contract
284
296
  * @param {string} runDir
285
297
  * @param {Map<string, NodeSnapshot>} states
286
- * @param {Map<string, Job>} running
298
+ * @param {Map<string, Job>} closed
287
299
  * @param {LockHandle} lock
288
300
  * @param {string} campaignPath
301
+ * @param {Map<string, Job>} running
289
302
  * @returns {Promise<void>}
290
303
  */
291
- export async function finalizeClosedJobs(contract, runDir, states, running, lock, campaignPath) {
304
+ export async function finalizeClosedJobs(contract, runDir, states, closed, lock, campaignPath, running) {
292
305
  // Advisory spend lines are checked every tick, before outcome handling: a
293
306
  // crossing must be visible while the spend is happening, not only when the
294
307
  // run is already over. The check never stops or transitions a node.
295
308
  await emitNodeAdvisories(contract, runDir, states);
296
- for (const [nodeId, job] of running) {
309
+ for (const [nodeId, job] of closed) {
297
310
  if (!job.closed || invocationAlive(job.invocation)) continue;
298
- running.delete(nodeId);
311
+ closed.delete(nodeId);
299
312
  const state = states.get(nodeId);
300
313
  if (!state) continue;
301
314
  // Usage is extracted and persisted BEFORE any outcome-specific handling:
@@ -9,6 +9,14 @@
9
9
  * break syntax and the sample is a deterministic eight, because a gate that
10
10
  * fails at random, or on a mutant that does not compile, is worse than no gate.
11
11
  *
12
+ * The kill fraction is declared by tier, not hand-picked per entry: the entry
13
+ * declares a risk tier and `MUTATION_TIERS` (`contract/verification.mjs`)
14
+ * fixes what each tier demands, so two nodes at the same tier sit the same
15
+ * bar. The whole entry — baseline plus sample — is held inside
16
+ * `MUTATION_TIME_BUDGET_MS`, and mutants the budget cannot afford stay in the
17
+ * denominator, so a suite too slow to sample inside the budget fails rather
18
+ * than passing on a narrowed set.
19
+ *
12
20
  * It is a separate module from `run-command.mjs` so that the "doing" of a
13
21
  * verification run and the "which file to break" policy do not grow together;
14
22
  * the runner receives the argv executor as a callback rather than importing it,
@@ -16,10 +24,27 @@
16
24
  */
17
25
  import { existsSync, readFileSync, statSync, writeFileSync } from "node:fs";
18
26
  import { resolve } from "node:path";
27
+ import { MUTATION_TIERS } from "../contract/verification.mjs";
19
28
 
20
29
  /** At most this many mutants run for one verification entry. */
21
30
  export const MUTATION_BUDGET = 8;
22
31
 
32
+ /**
33
+ * Wall-clock budget for one mutation entry — baseline plus every mutant run —
34
+ * in milliseconds. Each mutant re-runs the same argv, so each one costs what
35
+ * the baseline cost, and the runner refuses to start a mutant the remaining
36
+ * budget cannot afford.
37
+ *
38
+ * measured 2026-09-21 on this repository: the scenarios in
39
+ * test/engine/mutation.test.mjs cost 0.08–0.37s per run, so a node-sized entry
40
+ * (baseline plus eight mutants over a targeted suite) costs ~1–4s and samples
41
+ * in full. The budget admits any suite up to ~3.3s per run and denies slower
42
+ * ones every mutant — measured counter-example: `node --test test/contract`
43
+ * costs ~20s per run, and a mutation entry is scoped to the node's own
44
+ * verification command, not to a whole suite directory.
45
+ */
46
+ export const MUTATION_TIME_BUDGET_MS = 30_000;
47
+
23
48
  /**
24
49
  * The six operators, and only these: comparison and logical swaps. Each swap
25
50
  * keeps the expression well formed, so a surviving mutant is signal about the
@@ -109,18 +134,31 @@ function evenlySpaced(candidates, budget) {
109
134
  */
110
135
  export async function runMutation(command, baseCwd, options) {
111
136
  const mutation = command.mutation;
112
- if (!mutation) throw new TypeError("mutation runner requires a command.mutation threshold");
113
- const threshold = mutation.threshold;
114
- const selected = evenlySpaced(mutationCandidates(baseCwd, options.writeFiles ?? []), MUTATION_BUDGET);
137
+ if (!mutation) throw new TypeError("mutation runner requires a command.mutation risk tier");
138
+ const threshold = MUTATION_TIERS[mutation.tier];
139
+ const sample = evenlySpaced(mutationCandidates(baseCwd, options.writeFiles ?? []), MUTATION_BUDGET);
115
140
  /** @type {VerificationAttemptResult[]} */
116
141
  const attempts = [];
117
142
  const baseline = await options.run(1);
118
143
  attempts.push(baseline);
119
- if (!baseline.passed) return { passed: false, killed: 0, total: selected.length, threshold, attempts };
144
+ if (!baseline.passed) return { passed: false, killed: 0, total: sample.length, threshold, attempts };
145
+ // Every mutant re-runs the same argv, so the baseline's own measured duration
146
+ // is what one more attempt costs, and the budget is spent in those units. An
147
+ // executor that reports no duration (the unit-test seam) is charged zero and
148
+ // is never denied a mutant.
149
+ const attemptMs = Number.isFinite(baseline.durationMs) ? /** @type {number} */ (baseline.durationMs) : 0;
120
150
  /** @type {Map<string, string>} */
121
151
  const originals = new Map();
122
152
  let killed = 0;
123
- for (const [index, candidate] of selected.entries()) {
153
+ let spentMs = attemptMs;
154
+ let attempt = 1;
155
+ for (const candidate of sample) {
156
+ // Mutants the budget cannot afford are not silently dropped from the
157
+ // denominator: `total` stays the full sample, so truncation can only lower
158
+ // the kill fraction. A suite too slow to sample inside the budget fails a
159
+ // gate it is too slow to sit for — the narrowing the budget exists to
160
+ // prevent.
161
+ if (attemptMs > 0 && spentMs + attemptMs > MUTATION_TIME_BUDGET_MS) break;
124
162
  let original = originals.get(candidate.absolute);
125
163
  if (original === undefined) {
126
164
  original = readFileSync(candidate.absolute, "utf8");
@@ -129,16 +167,18 @@ export async function runMutation(command, baseCwd, options) {
129
167
  const mutated = `${original.slice(0, candidate.offset)}${candidate.to}${original.slice(candidate.offset + candidate.from.length)}`;
130
168
  /** @type {VerificationAttemptResult} */
131
169
  let result;
170
+ attempt += 1;
132
171
  try {
133
172
  writeFileSync(candidate.absolute, mutated);
134
- result = await options.run(index + 2);
173
+ result = await options.run(attempt);
135
174
  } finally {
136
175
  writeFileSync(candidate.absolute, original);
137
176
  }
138
177
  attempts.push(result);
178
+ spentMs += Number.isFinite(result.durationMs) ? /** @type {number} */ (result.durationMs) : 0;
139
179
  if (!result.passed) killed += 1;
140
180
  }
141
- const total = selected.length;
181
+ const total = sample.length;
142
182
  // No operators to break is a vacuous pass: there is no mutant the suite could
143
183
  // have failed to kill. The caller declared the entry, so an empty target is a
144
184
  // measurement of nothing, not a suite that proved nothing.
@@ -609,7 +609,7 @@ export async function recoverIntegrationTransactions(contract, runDir, states, l
609
609
  const node = contract.nodes.find((candidate) => candidate.id === transaction.node);
610
610
  const state = states.get(transaction.node);
611
611
  if (!node || !state || SETTLED.has(state.status)) return;
612
- const verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope);
612
+ const verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope);
613
613
  verdict.summary = "integrated candidate verification failed during recovery";
614
614
  applyRejection(contract, node, state, runDir, null, lock, states, campaignPath, verdict, {
615
615
  code: "verification_failed",
@@ -9,7 +9,14 @@ export { exhaustedUntilOf, normalizeProviderAvailability } from "../harnesses/in
9
9
  /** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
10
10
  /** @typedef {{harness?: string, model?: string, vendor: string, tier?: number|string, costRank?: number, [key: string]: unknown}} RuntimeLike */
11
11
  /** @typedef {{runtimes: Record<string, RuntimeLike>, runtimeDefaults?: {worker?: string, judge?: string}, nodes?: {id: string, runtime?: string, gate: {enabled: boolean, runtime?: string}}[]}} RuntimeContract */
12
- /** @typedef {{available: boolean, exhaustedUntil: string|null, reason: string}} RuntimeAvailability */
12
+ /**
13
+ * One runtime's catalogue record: what the harness de facto reported, and
14
+ * when. An unobservable datum is null -- never zero and never full allowance,
15
+ * so a runtime that reports nothing cannot look rested -- and an absent key on
16
+ * a record that predates the field reads as null at every reader. An
17
+ * observation older than its own window reads as unknown (`isRuntimeAvailable`).
18
+ * @typedef {{available: boolean, exhaustedUntil: string|null, reason: string, observedAt?: string|null, window?: string|null, remaining?: number|null}} RuntimeAvailability
19
+ */
13
20
  /** @typedef {{harness: string, model: string, vendor: string, tier: number, costRank: number, config?: Record<string, unknown>}} DiscoveryRuntime */
14
21
  /** @typedef {{id: string, runtime: RuntimeLike, order: number}} RuntimeCandidate */
15
22
  /** @typedef {import("../host/config.mjs").UserConfig} UserConfig */
@@ -89,7 +96,7 @@ export async function discoverRuntimes(runtimes, options = {}) {
89
96
  */
90
97
  export function availableCandidates(runtimes, availability = {}) {
91
98
  return Object.entries(runtimes)
92
- .filter(([id]) => isAvailable(availability[id]))
99
+ .filter(([id]) => isRuntimeAvailable(availability[id]))
93
100
  .map(([id, runtime], order) => ({ id, runtime, order }));
94
101
  }
95
102
 
@@ -172,7 +179,7 @@ export function nextSameTierRuntime(contract, stateRouting, role, current, attem
172
179
  const used = new Set(attempted);
173
180
  return Object.entries(contract.runtimes)
174
181
  .filter(([id, runtime]) => id !== current && !used.has(id) && sameTier(runtime, currentRuntime))
175
- .filter(([id]) => isAvailable(stateRouting.availability?.[id]))
182
+ .filter(([id]) => isRuntimeAvailable(stateRouting.availability?.[id]))
176
183
  .filter(([, runtime]) => role !== "judge" || runtime.vendor !== workerVendor)
177
184
  .sort((left, right) => runtimeOrder(left[1]) - runtimeOrder(right[1]))
178
185
  .map(([id]) => id)
@@ -226,10 +233,40 @@ function tierOrder(runtime) {
226
233
  return typeof runtime.tier === "number" ? runtime.tier : runtime.costRank ?? Number.MAX_SAFE_INTEGER;
227
234
  }
228
235
 
229
- /** @param {RuntimeAvailability|undefined} availability @returns {boolean} */
230
- function isAvailable(availability) {
236
+ /**
237
+ * Span in seconds of every rate-limit window label a harness reports. The
238
+ * labels are claude's `rateLimitType` values (measured 2026-09-17, the
239
+ * `rate_limit_event` line recorded in `src/harnesses/protocol.mjs`); a label
240
+ * missing here cannot prove staleness, so its observation never self-expires.
241
+ *
242
+ * @type {Readonly<Record<string, number>>}
243
+ */
244
+ const AVAILABILITY_WINDOW_SEC = Object.freeze({ five_hour: 5 * 3600, seven_day: 7 * 86400 });
245
+
246
+ /**
247
+ * May a runtime be admitted on this catalogue record? Exhaustion is waited
248
+ * out on `exhaustedUntil`; an observation older than its own window reads as
249
+ * unknown and admits nothing, because unknown must not look rested. This is
250
+ * the one home of the rule: plan routing and engine composition both read it,
251
+ * so the null and staleness semantics cannot drift between readers. The
252
+ * parameter is typed on the fields the rule reads, not on the full record --
253
+ * the plan's table copy names no `reason`.
254
+ *
255
+ * @param {{available: boolean, exhaustedUntil: string|null, observedAt?: string|null, window?: string|null, [key: string]: unknown}|undefined} availability
256
+ * @param {number} [now] epoch milliseconds; defaults to the current clock
257
+ * @returns {boolean}
258
+ */
259
+ export function isRuntimeAvailable(availability, now = Date.now()) {
231
260
  if (!availability) return false;
232
- if (availability.available === true) return !availability.exhaustedUntil || Date.parse(availability.exhaustedUntil) <= Date.now();
233
- return Boolean(availability.exhaustedUntil && Date.parse(availability.exhaustedUntil) <= Date.now());
261
+ const rested = availability.available === true
262
+ ? !availability.exhaustedUntil || Date.parse(availability.exhaustedUntil) <= now
263
+ : Boolean(availability.exhaustedUntil && Date.parse(availability.exhaustedUntil) <= now);
264
+ if (!rested) return false;
265
+ const windowSec = availability.window === undefined || availability.window === null
266
+ ? undefined
267
+ : AVAILABILITY_WINDOW_SEC[availability.window];
268
+ if (windowSec === undefined || availability.observedAt === undefined || availability.observedAt === null) return true;
269
+ const observedAt = Date.parse(availability.observedAt);
270
+ return !Number.isNaN(observedAt) && observedAt + windowSec * 1000 >= now;
234
271
  }
235
272
 
@@ -463,23 +463,23 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
463
463
  // batched call it used to be, but the chain still runs them one at a time.
464
464
  // A settlement may itself dispatch the node's next phase (a judge, a
465
465
  // revision) through the same `startJudge`/`startWorker` calls dispatch below
466
- // uses; those land in `slot`, not the real `running`, so they are copied
467
- // back into `running` in a `finally` -- unconditionally, win or lose, so a
468
- // job a settlement started before failing (a lost lock, a programmer error)
469
- // is still visible to the cleanup sweeps below rather than leaked. The
470
- // node's own steps stay ordered by never starting a second settlement for a
471
- // node whose first has not yet cleared `pendingSettlements`.
466
+ // uses, so it is handed the real `running` to dispatch into: a job it starts
467
+ // is counted against `maxParallel` from the instant the process exists, and
468
+ // `applyRejection` reads that same map to decide whether the run has a slot
469
+ // for the revision at all. It used to dispatch into the throwaway one-entry
470
+ // map instead, copied back only once the settlement returned, which is how
471
+ // run state-location-and-routing-economics-13 came to hold two workers under
472
+ // `maxParallel: 1` on 2026-09-21. What it may *settle* is still only its own
473
+ // node: the one-entry map below is the job, not the run. The node's own
474
+ // steps stay ordered by never starting a second settlement for a node whose
475
+ // first has not yet cleared `pendingSettlements`.
472
476
  const settleClosedJobsInBackground = () => {
473
477
  for (const [nodeId, job] of [...running]) {
474
478
  if (pendingSettlements.has(nodeId) || !job.closed || invocationAlive(job.invocation)) continue;
475
479
  running.delete(nodeId);
476
- const slot = new Map([[nodeId, job]]);
477
480
  const settlement = settlementQueue
478
- .then(() => finalizeClosedJobs(contract, runDir, states, slot, lock, campaign.path))
479
- .finally(() => {
480
- for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
481
- pendingSettlements.delete(nodeId);
482
- });
481
+ .then(() => finalizeClosedJobs(contract, runDir, states, new Map([[nodeId, job]]), lock, campaign.path, running))
482
+ .finally(() => pendingSettlements.delete(nodeId));
483
483
  // The queue itself must never reject -- a rejected settlement (a lost
484
484
  // lock, a programmer error) would otherwise wedge every node queued
485
485
  // behind it. The rejection still reaches whoever awaits the real
@@ -635,20 +635,16 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
635
635
  // does, on the same per-run candidate ref and worktree
636
636
  // `settlementQueue` exists to serialize -- so it is dispatched the
637
637
  // same way: chained onto the queue rather than awaited here, using
638
- // its own one-entry `slot` merged back into `running` in a
639
- // `finally`. The node's own order is untouched (still one entry at
640
- // a time, gated by `pendingSettlements`); only the tick stops
641
- // waiting behind it.
638
+ // `running` itself as its dispatch map, so the judge it starts is
639
+ // counted the instant it exists. The node's own order is untouched
640
+ // (still one entry at a time, gated by `pendingSettlements`); only
641
+ // the tick stops waiting behind it.
642
642
  if (state.phase === "judge" && state.result) {
643
643
  const workerResult = state.result;
644
- const slot = new Map();
645
644
  const settlement = settlementQueue
646
- .then(() => startJudge(contract, node, state, runDir, slot, workerResult, lock, states, campaign.path))
647
- .then((round) => applyJudgeRound(round, contract, node, state, runDir, slot, lock, states, campaign.path, workerResult))
648
- .finally(() => {
649
- for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
650
- pendingSettlements.delete(node.id);
651
- });
645
+ .then(() => startJudge(contract, node, state, runDir, running, workerResult, lock, states, campaign.path))
646
+ .then((round) => applyJudgeRound(round, contract, node, state, runDir, running, lock, states, campaign.path, workerResult))
647
+ .finally(() => pendingSettlements.delete(node.id));
652
648
  settlementQueue = settlement.catch(() => {});
653
649
  settlement.catch((error) => {
654
650
  if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
@@ -3,15 +3,19 @@
3
3
  * touched, and what to do when those differ.
4
4
  *
5
5
  * Scope is advisory by design -- an unexpected write is recorded as a finding
6
- * and shown to the judge, not treated as a crime -- with one exception:
6
+ * and shown to the judge, not treated as a crime -- with two exceptions:
7
7
  * `resolveUnknownEffect` decides whether an invocation whose effect is unproven
8
- * may be replayed at all, and a dirty scope there is a refusal.
8
+ * may be replayed at all, and a dirty scope there is a refusal; and a write
9
+ * that lands on a file the node's own proof names is never deferred, because a
10
+ * verification that passes over an edited prover has proven nothing.
9
11
  */
12
+ import { relative, resolve } from "node:path";
13
+
10
14
  import { SETTLED } from "./prompts.mjs";
11
15
  import { appendTransitionEvent, recordExecutionOverride, transition, writeNode } from "./state.mjs";
12
16
  import { attemptWorkspace } from "../repo/worktree.mjs";
13
17
 
14
- import { errorCode, errorMessage, excerpt } from "../util.mjs";
18
+ import { errorCode, errorMessage, excerpt, isContained } from "../util.mjs";
15
19
  import { executeControllerVerification } from "./verify.mjs";
16
20
  import { providerReceiptsFromInvocationTail, settleInvocation } from "../run/operations.mjs";
17
21
  import { readJson } from "../run/store.mjs";
@@ -98,6 +102,94 @@ export function workerScope(taskPacket) {
98
102
  roots: taskPacket.writeRoots ?? [],
99
103
  };
100
104
  }
105
+ /**
106
+ * @typedef {{tokens: string[], cwd: string, literal: boolean, citation: string}} ProofCitation
107
+ */
108
+
109
+ /**
110
+ * Everything a node's own proofs name: a Definition of Done `path` proof's
111
+ * path, the words of a `command` proof (a command proof carries a display
112
+ * string, not an argv, so a quoted path holding a space is not recovered), the
113
+ * argv of the verification entry a `verification` proof references, and the
114
+ * argv of every verification command the packet declares.
115
+ *
116
+ * @param {ValidatedNode} node
117
+ * @returns {ProofCitation[]}
118
+ */
119
+ function proofCitations(node) {
120
+ const commands = node.taskPacket.verification ?? [];
121
+ /** @type {ProofCitation[]} */
122
+ const citations = [];
123
+ // Definition of Done items come first so that a path both a checklist item
124
+ // and a verification command name is reported under the checklist item, the
125
+ // name a human reading the failure can act on.
126
+ for (const item of node.definitionOfDone ?? []) {
127
+ const proof = item.proof;
128
+ if (!proof) continue;
129
+ if (proof.kind === "path") {
130
+ citations.push({ tokens: [proof.ref], cwd: ".", literal: true, citation: `${item.id} path proof` });
131
+ } else if (proof.kind === "command") {
132
+ citations.push({ tokens: proof.ref.split(/\s+/u), cwd: ".", literal: false, citation: `${item.id} command proof` });
133
+ } else {
134
+ const command = commands[Number.parseInt(proof.ref, 10)];
135
+ if (command) citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `${item.id} verification[${proof.ref}] proof` });
136
+ }
137
+ }
138
+ for (const [index, command] of commands.entries()) {
139
+ citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `verification[${index}]` });
140
+ }
141
+ return citations;
142
+ }
143
+ const PATH_SEPARATOR = /[\\/]/u;
144
+
145
+ /**
146
+ * The unexpected writes that landed on a file the node's own proof names.
147
+ * Matching a command's argv against files is inherently approximate, so this is
148
+ * lexical and deliberately narrow. A word is read as a path only when it is not
149
+ * an option, is shaped like one (a `path` proof's ref, or a word carrying a
150
+ * separator or an extension), and resolves inside the workspace; it then claims
151
+ * an unexpected path it equals, or -- when it carries a separator or is a `path`
152
+ * proof's ref, so a directory really was named -- one it is the directory
153
+ * prefix of.
154
+ *
155
+ * What it deliberately does not catch: a file a proof reaches through a script
156
+ * (`npm test`), a shell string, or a glob the tool expands itself. The bare
157
+ * words of a command are never paths, which is what keeps the ordinary case
158
+ * advisory -- measured against the real node `requirement-ids-reach-the-node`
159
+ * (run `state-location-and-routing-economics-10-requirement-ids-and-closure`),
160
+ * whose legitimate out-of-scope write to `src/plan/freeze.mjs` is claimed by
161
+ * none of its proofs: not by `npm run typecheck`, not by the two test files its
162
+ * `command` proofs name, and not by the loose words of a quoted
163
+ * `--test-name-pattern`.
164
+ *
165
+ * @param {ValidatedNode} node
166
+ * @param {string[]} unexpectedPaths
167
+ * @param {string} workspace
168
+ * @returns {{path: string, citation: string}[]}
169
+ */
170
+ function proofCitedWrites(node, unexpectedPaths, workspace) {
171
+ /** @type {Map<string, string>} */
172
+ const cited = new Map();
173
+ for (const citation of proofCitations(node)) {
174
+ const base = resolve(workspace, citation.cwd);
175
+ for (const token of citation.tokens) {
176
+ if (!token || token.startsWith("-")) continue;
177
+ const directory = citation.literal || PATH_SEPARATOR.test(token);
178
+ if (!directory && !/\.[A-Za-z0-9]+$/u.test(token)) continue;
179
+ const target = resolve(base, token);
180
+ if (!isContained(workspace, target)) continue;
181
+ const named = relative(workspace, target).replaceAll("\\", "/");
182
+ // The workspace root itself names no file in particular: a proof run from
183
+ // the root must not make every unexpected write a proof-citing one.
184
+ if (!named) continue;
185
+ for (const path of unexpectedPaths) {
186
+ if (cited.has(path)) continue;
187
+ if (path === named || (directory && path.startsWith(`${named}/`))) cited.set(path, citation.citation);
188
+ }
189
+ }
190
+ }
191
+ return [...cited].map(([path, citation]) => ({ path, citation }));
192
+ }
101
193
  /**
102
194
  * @param {ValidatedContract} contract
103
195
  * @param {string} runDir
@@ -122,9 +214,19 @@ export function checkWorkerScope(contract, runDir, job, lock, options = {}) {
122
214
  // A completed attempt whose controller verification passes never fails
123
215
  // on scope alone (TECH-SPEC lean, rule 1): the caller defers the verdict
124
216
  // until verification has run and records an advisory finding instead.
125
- if (options.deferViolation) return true;
126
- const shown = bounded.unexpectedPaths.slice(0, 8).join(", ");
127
- const message = `unexpected paths changed (${scope.unexpectedPaths.length}): ${shown}`;
217
+ // The single exception is a write onto a file the node's own proof names:
218
+ // verification then passes because the attempt edited the thing doing the
219
+ // proving, and an advisory nobody must read before the gate is too weak a
220
+ // signal for that. Scanned over the bounded path list -- the same first 64
221
+ // paths every other surface reports.
222
+ const cited = proofCitedWrites(job.node, bounded.unexpectedPaths, job.cwd);
223
+ if (options.deferViolation && !cited.length) return true;
224
+ const message = cited.length
225
+ // Kept short on purpose: an error message is capped at 120 characters,
226
+ // and the proof that names the path is the part a reader cannot recover
227
+ // from `state.scope` afterwards.
228
+ ? `proof-cited unexpected write (${cited.length}): ${cited.slice(0, 8).map(({ path, citation }) => `${path} (${citation})`).join(", ")}`
229
+ : `unexpected paths changed (${scope.unexpectedPaths.length}): ${bounded.unexpectedPaths.slice(0, 8).join(", ")}`;
128
230
  if (!SETTLED.has(state.status)) {
129
231
  transition(runDir, state, "failed", { phase: "worker", error: { code: "unexpected_write", message: excerpt(message) } }, lock);
130
232
  appendTransitionEvent(runDir, state, "failed", "failed", {
@@ -69,18 +69,29 @@ export function applyRejection(contract, node, state, runDir, running, lock, sta
69
69
  // dispatch below and a later scheduler dispatch (the `running` is null
70
70
  // path) carry it.
71
71
  state.previousAttempt = renderPreviousAttemptSection(state) ?? state.previousAttempt;
72
- if (running) {
72
+ // A revision is a new attempt, and a new attempt is a dispatch: it starts
73
+ // here only against a slot the run actually has free. `running` is the
74
+ // scheduler's own map, with this node's own closed job already out of it,
75
+ // so `running.size` is exactly what the dispatch loop's own
76
+ // `maxParallel - running.size` will read. A settlement that dispatched
77
+ // regardless is how run state-location-and-routing-economics-13 ran two
78
+ // workers under `maxParallel: 1` on 2026-09-21, and two of that day's
79
+ // three OOM kills happened with more running than the contract declared.
80
+ if (running && running.size < contract.maxParallel) {
73
81
  // Dispatching here owns the increment, because `startWorker` expects the
74
82
  // attempt number it is about to run under.
75
83
  state.attempt += 1;
76
84
  startWorker(contract, node, state, runDir, running, retryPrompt(node, verdict), lock, states, campaignPath, { forceFresh });
77
85
  return;
78
86
  }
79
- // Handing the node back to the scheduler instead: its dispatch increments
80
- // on the way out, so incrementing here too spent two attempt numbers on one
81
- // retry. Observed 2026-09-13 on a resume after a killed controller — a node
82
- // that ran twice reported attempt 3, with no `…2.*` logs and a
83
- // `worktree.previousAttempt` naming an attempt that never existed.
87
+ // Handing the node back to the scheduler instead -- because there is no
88
+ // loop to hand it to (recovery, resume), or because the run is full and
89
+ // the scheduler is the one place that knows when it stops being full. Its
90
+ // dispatch increments on the way out, so incrementing here too spent two
91
+ // attempt numbers on one retry. Observed 2026-09-13 on a resume after a
92
+ // killed controller — a node that ran twice reported attempt 3, with no
93
+ // `…2.*` logs and a `worktree.previousAttempt` naming an attempt that
94
+ // never existed.
84
95
  transition(runDir, state, "pending", { phase: "worker", error: null }, lock);
85
96
  return;
86
97
  }
@@ -91,7 +102,7 @@ export function applyRejection(contract, node, state, runDir, running, lock, sta
91
102
  }, lock);
92
103
  }
93
104
  /** Deterministic verification failure settles through the shared rejection path. The verdict carries this attempt's unexpected paths, so a red attempt reports them whether it stops here or starts its revision (TECH-SPEC lean, rule 1). @param {ValidatedContract} contract @param {ValidatedNode} node @param {NodeSnapshot} state @param {string} runDir @param {Map<string, Job>|null} running @param {LockHandle} lock @param {Map<string, NodeSnapshot>} states @param {string} campaignPath @param {JudgeVerdict} [verdict] */
94
- export function applyVerificationFailure(contract, node, state, runDir, running, lock, states, campaignPath, verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope)) {
105
+ export function applyVerificationFailure(contract, node, state, runDir, running, lock, states, campaignPath, verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope)) {
95
106
  applyRejection(contract, node, state, runDir, running, lock, states, campaignPath, verdict, { code: "verification_failed", label: "verification" });
96
107
  }
97
108
 
@@ -193,7 +204,7 @@ export async function settleDone(contract, node, state, runDir, lock, states, ca
193
204
  removeWorktree(contract.cwd, acceptedPath);
194
205
  },
195
206
  onVerificationFailure: async (transaction) => {
196
- const verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope);
207
+ const verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope);
197
208
  verdict.summary = "integrated candidate verification failed";
198
209
  const divergent = candidateOnlyFailures(state.verification, transaction.candidateEvidence);
199
210
  verdict.findings = [...(verdict.findings ?? []), {
@@ -27,6 +27,7 @@ import { errorMessage } from "../util.mjs";
27
27
  import { boundedGitSync } from "../repo/worktree.mjs";
28
28
  import { routeRuntime } from "../contract/runtime.mjs";
29
29
  import { NOTIFY_BIN_ENV, noTransportWarning } from "../notify/index.mjs";
30
+ import { NOTIFY_SESSION_ENV, sessionWakeNotice } from "../notify/session.mjs";
30
31
  import { findExecutable } from "./platform.mjs";
31
32
  import { colorLevel, statusToken } from "../cli/brand.mjs";
32
33
  import { RUNS_DIR_NAME } from "../run/paths.mjs";
@@ -209,9 +210,9 @@ export function environmentPreflight(options) {
209
210
  */
210
211
  export function notifyTransportCheck(env = process.env) {
211
212
  const warning = noTransportWarning(env);
212
- return warning
213
- ? fail("notify transport", warning, true)
214
- : pass("notify transport", `${NOTIFY_BIN_ENV}=${env[NOTIFY_BIN_ENV]}`);
213
+ if (warning) return fail("notify transport", warning, true);
214
+ const external = env[NOTIFY_BIN_ENV] ? `${NOTIFY_BIN_ENV}=${env[NOTIFY_BIN_ENV]}` : `${NOTIFY_BIN_ENV} unset`;
215
+ return pass("notify transport", `${external} · ${sessionWakeNotice(env)}`);
215
216
  }
216
217
 
217
218
  /** @param {EnvReport} report @returns {EnvCheck[]} the checks that block a dispatch */
@@ -265,6 +266,7 @@ export function declaredVerificationCommands(contract) {
265
266
  */
266
267
  const SIDE_EFFECT_ENV_KEYS = [
267
268
  NOTIFY_BIN_ENV, // a measurement must not notify a human
269
+ NOTIFY_SESSION_ENV, // nor wake the harness session it was measured from
268
270
  "FABERUN_CODEX_BIN", // could redirect the timed command at a live, paid codex binary instead of this repository's own fixtures
269
271
  "FABERUN_CLAUDE_BIN", // same, for the claude harness
270
272
  "FABERUN_AGY_BIN", // same, for the agy harness