faberun 0.14.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/faberun/references/operations.md +22 -11
- package/src/cli/plan.mjs +40 -1
- package/src/cli.mjs +1 -0
- package/src/contract/runtime.mjs +50 -2
- package/src/contract/verification.mjs +26 -10
- package/src/engine/assignment.mjs +33 -6
- package/src/engine/backoff.mjs +11 -5
- package/src/engine/cancel.mjs +20 -6
- package/src/engine/failover.mjs +62 -1
- package/src/engine/judge-gate.mjs +63 -5
- package/src/engine/lifecycle.mjs +17 -4
- package/src/engine/mutation.mjs +47 -7
- package/src/engine/resume.mjs +1 -1
- package/src/engine/runtime-discovery.mjs +44 -7
- package/src/engine/scheduler.mjs +19 -23
- package/src/engine/scope.mjs +108 -6
- package/src/engine/settle.mjs +19 -8
- package/src/host/preflight.mjs +5 -3
- package/src/notify/index.mjs +66 -32
- package/src/notify/session.mjs +240 -0
- package/src/plan/pipeline.mjs +215 -22
- package/src/plan/repo-facts.mjs +2 -0
- package/src/plan/routing.mjs +151 -33
- package/src/plan/sizing.mjs +33 -0
- package/src/repo/worktree.mjs +28 -0
- package/src/report/final.mjs +2 -1
- package/src/report/locale.mjs +164 -0
- package/src/report/message.mjs +602 -0
- package/src/report/packet-repetition.mjs +121 -0
- package/src/report/progress.mjs +33 -153
- package/src/report/render.mjs +3 -3
- package/src/seat/tmux.mjs +12 -3
package/src/engine/lifecycle.mjs
CHANGED
|
@@ -280,22 +280,35 @@ export function clearTierExhaustion(state) {
|
|
|
280
280
|
}
|
|
281
281
|
|
|
282
282
|
/**
|
|
283
|
+
* Settle what a closed invocation produced, and start whatever the outcome
|
|
284
|
+
* earns next.
|
|
285
|
+
*
|
|
286
|
+
* The two maps are two different things, and conflating them is what let a run
|
|
287
|
+
* exceed its own `maxParallel` (measured 2026-09-21, run
|
|
288
|
+
* state-location-and-routing-economics-13): `closed` is the job this call owns
|
|
289
|
+
* -- the scheduler hands one node's job at a time so two settlements never
|
|
290
|
+
* interleave -- while `running` is the run's live dispatch authority, the very
|
|
291
|
+
* map the scheduler's `maxParallel - running.size` counts. Anything started
|
|
292
|
+
* here lands in `running` and is accounted from that instant; nothing started
|
|
293
|
+
* here may be settled by this call.
|
|
294
|
+
*
|
|
283
295
|
* @param {ValidatedContract} contract
|
|
284
296
|
* @param {string} runDir
|
|
285
297
|
* @param {Map<string, NodeSnapshot>} states
|
|
286
|
-
* @param {Map<string, Job>}
|
|
298
|
+
* @param {Map<string, Job>} closed
|
|
287
299
|
* @param {LockHandle} lock
|
|
288
300
|
* @param {string} campaignPath
|
|
301
|
+
* @param {Map<string, Job>} running
|
|
289
302
|
* @returns {Promise<void>}
|
|
290
303
|
*/
|
|
291
|
-
export async function finalizeClosedJobs(contract, runDir, states,
|
|
304
|
+
export async function finalizeClosedJobs(contract, runDir, states, closed, lock, campaignPath, running) {
|
|
292
305
|
// Advisory spend lines are checked every tick, before outcome handling: a
|
|
293
306
|
// crossing must be visible while the spend is happening, not only when the
|
|
294
307
|
// run is already over. The check never stops or transitions a node.
|
|
295
308
|
await emitNodeAdvisories(contract, runDir, states);
|
|
296
|
-
for (const [nodeId, job] of
|
|
309
|
+
for (const [nodeId, job] of closed) {
|
|
297
310
|
if (!job.closed || invocationAlive(job.invocation)) continue;
|
|
298
|
-
|
|
311
|
+
closed.delete(nodeId);
|
|
299
312
|
const state = states.get(nodeId);
|
|
300
313
|
if (!state) continue;
|
|
301
314
|
// Usage is extracted and persisted BEFORE any outcome-specific handling:
|
package/src/engine/mutation.mjs
CHANGED
|
@@ -9,6 +9,14 @@
|
|
|
9
9
|
* break syntax and the sample is a deterministic eight, because a gate that
|
|
10
10
|
* fails at random, or on a mutant that does not compile, is worse than no gate.
|
|
11
11
|
*
|
|
12
|
+
* The kill fraction is declared by tier, not hand-picked per entry: the entry
|
|
13
|
+
* declares a risk tier and `MUTATION_TIERS` (`contract/verification.mjs`)
|
|
14
|
+
* fixes what each tier demands, so two nodes at the same tier sit the same
|
|
15
|
+
* bar. The whole entry — baseline plus sample — is held inside
|
|
16
|
+
* `MUTATION_TIME_BUDGET_MS`, and mutants the budget cannot afford stay in the
|
|
17
|
+
* denominator, so a suite too slow to sample inside the budget fails rather
|
|
18
|
+
* than passing on a narrowed set.
|
|
19
|
+
*
|
|
12
20
|
* It is a separate module from `run-command.mjs` so that the "doing" of a
|
|
13
21
|
* verification run and the "which file to break" policy do not grow together;
|
|
14
22
|
* the runner receives the argv executor as a callback rather than importing it,
|
|
@@ -16,10 +24,27 @@
|
|
|
16
24
|
*/
|
|
17
25
|
import { existsSync, readFileSync, statSync, writeFileSync } from "node:fs";
|
|
18
26
|
import { resolve } from "node:path";
|
|
27
|
+
import { MUTATION_TIERS } from "../contract/verification.mjs";
|
|
19
28
|
|
|
20
29
|
/** At most this many mutants run for one verification entry. */
|
|
21
30
|
export const MUTATION_BUDGET = 8;
|
|
22
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Wall-clock budget for one mutation entry — baseline plus every mutant run —
|
|
34
|
+
* in milliseconds. Each mutant re-runs the same argv, so each one costs what
|
|
35
|
+
* the baseline cost, and the runner refuses to start a mutant the remaining
|
|
36
|
+
* budget cannot afford.
|
|
37
|
+
*
|
|
38
|
+
* measured 2026-09-21 on this repository: the scenarios in
|
|
39
|
+
* test/engine/mutation.test.mjs cost 0.08–0.37s per run, so a node-sized entry
|
|
40
|
+
* (baseline plus eight mutants over a targeted suite) costs ~1–4s and samples
|
|
41
|
+
* in full. The budget admits any suite up to ~3.3s per run and denies slower
|
|
42
|
+
* ones every mutant — measured counter-example: `node --test test/contract`
|
|
43
|
+
* costs ~20s per run, and a mutation entry is scoped to the node's own
|
|
44
|
+
* verification command, not to a whole suite directory.
|
|
45
|
+
*/
|
|
46
|
+
export const MUTATION_TIME_BUDGET_MS = 30_000;
|
|
47
|
+
|
|
23
48
|
/**
|
|
24
49
|
* The six operators, and only these: comparison and logical swaps. Each swap
|
|
25
50
|
* keeps the expression well formed, so a surviving mutant is signal about the
|
|
@@ -109,18 +134,31 @@ function evenlySpaced(candidates, budget) {
|
|
|
109
134
|
*/
|
|
110
135
|
export async function runMutation(command, baseCwd, options) {
|
|
111
136
|
const mutation = command.mutation;
|
|
112
|
-
if (!mutation) throw new TypeError("mutation runner requires a command.mutation
|
|
113
|
-
const threshold = mutation.
|
|
114
|
-
const
|
|
137
|
+
if (!mutation) throw new TypeError("mutation runner requires a command.mutation risk tier");
|
|
138
|
+
const threshold = MUTATION_TIERS[mutation.tier];
|
|
139
|
+
const sample = evenlySpaced(mutationCandidates(baseCwd, options.writeFiles ?? []), MUTATION_BUDGET);
|
|
115
140
|
/** @type {VerificationAttemptResult[]} */
|
|
116
141
|
const attempts = [];
|
|
117
142
|
const baseline = await options.run(1);
|
|
118
143
|
attempts.push(baseline);
|
|
119
|
-
if (!baseline.passed) return { passed: false, killed: 0, total:
|
|
144
|
+
if (!baseline.passed) return { passed: false, killed: 0, total: sample.length, threshold, attempts };
|
|
145
|
+
// Every mutant re-runs the same argv, so the baseline's own measured duration
|
|
146
|
+
// is what one more attempt costs, and the budget is spent in those units. An
|
|
147
|
+
// executor that reports no duration (the unit-test seam) is charged zero and
|
|
148
|
+
// is never denied a mutant.
|
|
149
|
+
const attemptMs = Number.isFinite(baseline.durationMs) ? /** @type {number} */ (baseline.durationMs) : 0;
|
|
120
150
|
/** @type {Map<string, string>} */
|
|
121
151
|
const originals = new Map();
|
|
122
152
|
let killed = 0;
|
|
123
|
-
|
|
153
|
+
let spentMs = attemptMs;
|
|
154
|
+
let attempt = 1;
|
|
155
|
+
for (const candidate of sample) {
|
|
156
|
+
// Mutants the budget cannot afford are not silently dropped from the
|
|
157
|
+
// denominator: `total` stays the full sample, so truncation can only lower
|
|
158
|
+
// the kill fraction. A suite too slow to sample inside the budget fails a
|
|
159
|
+
// gate it is too slow to sit for — the narrowing the budget exists to
|
|
160
|
+
// prevent.
|
|
161
|
+
if (attemptMs > 0 && spentMs + attemptMs > MUTATION_TIME_BUDGET_MS) break;
|
|
124
162
|
let original = originals.get(candidate.absolute);
|
|
125
163
|
if (original === undefined) {
|
|
126
164
|
original = readFileSync(candidate.absolute, "utf8");
|
|
@@ -129,16 +167,18 @@ export async function runMutation(command, baseCwd, options) {
|
|
|
129
167
|
const mutated = `${original.slice(0, candidate.offset)}${candidate.to}${original.slice(candidate.offset + candidate.from.length)}`;
|
|
130
168
|
/** @type {VerificationAttemptResult} */
|
|
131
169
|
let result;
|
|
170
|
+
attempt += 1;
|
|
132
171
|
try {
|
|
133
172
|
writeFileSync(candidate.absolute, mutated);
|
|
134
|
-
result = await options.run(
|
|
173
|
+
result = await options.run(attempt);
|
|
135
174
|
} finally {
|
|
136
175
|
writeFileSync(candidate.absolute, original);
|
|
137
176
|
}
|
|
138
177
|
attempts.push(result);
|
|
178
|
+
spentMs += Number.isFinite(result.durationMs) ? /** @type {number} */ (result.durationMs) : 0;
|
|
139
179
|
if (!result.passed) killed += 1;
|
|
140
180
|
}
|
|
141
|
-
const total =
|
|
181
|
+
const total = sample.length;
|
|
142
182
|
// No operators to break is a vacuous pass: there is no mutant the suite could
|
|
143
183
|
// have failed to kill. The caller declared the entry, so an empty target is a
|
|
144
184
|
// measurement of nothing, not a suite that proved nothing.
|
package/src/engine/resume.mjs
CHANGED
|
@@ -609,7 +609,7 @@ export async function recoverIntegrationTransactions(contract, runDir, states, l
|
|
|
609
609
|
const node = contract.nodes.find((candidate) => candidate.id === transaction.node);
|
|
610
610
|
const state = states.get(transaction.node);
|
|
611
611
|
if (!node || !state || SETTLED.has(state.status)) return;
|
|
612
|
-
const verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope);
|
|
612
|
+
const verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope);
|
|
613
613
|
verdict.summary = "integrated candidate verification failed during recovery";
|
|
614
614
|
applyRejection(contract, node, state, runDir, null, lock, states, campaignPath, verdict, {
|
|
615
615
|
code: "verification_failed",
|
|
@@ -9,7 +9,14 @@ export { exhaustedUntilOf, normalizeProviderAvailability } from "../harnesses/in
|
|
|
9
9
|
/** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
|
|
10
10
|
/** @typedef {{harness?: string, model?: string, vendor: string, tier?: number|string, costRank?: number, [key: string]: unknown}} RuntimeLike */
|
|
11
11
|
/** @typedef {{runtimes: Record<string, RuntimeLike>, runtimeDefaults?: {worker?: string, judge?: string}, nodes?: {id: string, runtime?: string, gate: {enabled: boolean, runtime?: string}}[]}} RuntimeContract */
|
|
12
|
-
/**
|
|
12
|
+
/**
|
|
13
|
+
* One runtime's catalogue record: what the harness de facto reported, and
|
|
14
|
+
* when. An unobservable datum is null -- never zero and never full allowance,
|
|
15
|
+
* so a runtime that reports nothing cannot look rested -- and an absent key on
|
|
16
|
+
* a record that predates the field reads as null at every reader. An
|
|
17
|
+
* observation older than its own window reads as unknown (`isRuntimeAvailable`).
|
|
18
|
+
* @typedef {{available: boolean, exhaustedUntil: string|null, reason: string, observedAt?: string|null, window?: string|null, remaining?: number|null}} RuntimeAvailability
|
|
19
|
+
*/
|
|
13
20
|
/** @typedef {{harness: string, model: string, vendor: string, tier: number, costRank: number, config?: Record<string, unknown>}} DiscoveryRuntime */
|
|
14
21
|
/** @typedef {{id: string, runtime: RuntimeLike, order: number}} RuntimeCandidate */
|
|
15
22
|
/** @typedef {import("../host/config.mjs").UserConfig} UserConfig */
|
|
@@ -89,7 +96,7 @@ export async function discoverRuntimes(runtimes, options = {}) {
|
|
|
89
96
|
*/
|
|
90
97
|
export function availableCandidates(runtimes, availability = {}) {
|
|
91
98
|
return Object.entries(runtimes)
|
|
92
|
-
.filter(([id]) =>
|
|
99
|
+
.filter(([id]) => isRuntimeAvailable(availability[id]))
|
|
93
100
|
.map(([id, runtime], order) => ({ id, runtime, order }));
|
|
94
101
|
}
|
|
95
102
|
|
|
@@ -172,7 +179,7 @@ export function nextSameTierRuntime(contract, stateRouting, role, current, attem
|
|
|
172
179
|
const used = new Set(attempted);
|
|
173
180
|
return Object.entries(contract.runtimes)
|
|
174
181
|
.filter(([id, runtime]) => id !== current && !used.has(id) && sameTier(runtime, currentRuntime))
|
|
175
|
-
.filter(([id]) =>
|
|
182
|
+
.filter(([id]) => isRuntimeAvailable(stateRouting.availability?.[id]))
|
|
176
183
|
.filter(([, runtime]) => role !== "judge" || runtime.vendor !== workerVendor)
|
|
177
184
|
.sort((left, right) => runtimeOrder(left[1]) - runtimeOrder(right[1]))
|
|
178
185
|
.map(([id]) => id)
|
|
@@ -226,10 +233,40 @@ function tierOrder(runtime) {
|
|
|
226
233
|
return typeof runtime.tier === "number" ? runtime.tier : runtime.costRank ?? Number.MAX_SAFE_INTEGER;
|
|
227
234
|
}
|
|
228
235
|
|
|
229
|
-
/**
|
|
230
|
-
|
|
236
|
+
/**
|
|
237
|
+
* Span in seconds of every rate-limit window label a harness reports. The
|
|
238
|
+
* labels are claude's `rateLimitType` values (measured 2026-09-17, the
|
|
239
|
+
* `rate_limit_event` line recorded in `src/harnesses/protocol.mjs`); a label
|
|
240
|
+
* missing here cannot prove staleness, so its observation never self-expires.
|
|
241
|
+
*
|
|
242
|
+
* @type {Readonly<Record<string, number>>}
|
|
243
|
+
*/
|
|
244
|
+
const AVAILABILITY_WINDOW_SEC = Object.freeze({ five_hour: 5 * 3600, seven_day: 7 * 86400 });
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* May a runtime be admitted on this catalogue record? Exhaustion is waited
|
|
248
|
+
* out on `exhaustedUntil`; an observation older than its own window reads as
|
|
249
|
+
* unknown and admits nothing, because unknown must not look rested. This is
|
|
250
|
+
* the one home of the rule: plan routing and engine composition both read it,
|
|
251
|
+
* so the null and staleness semantics cannot drift between readers. The
|
|
252
|
+
* parameter is typed on the fields the rule reads, not on the full record --
|
|
253
|
+
* the plan's table copy names no `reason`.
|
|
254
|
+
*
|
|
255
|
+
* @param {{available: boolean, exhaustedUntil: string|null, observedAt?: string|null, window?: string|null, [key: string]: unknown}|undefined} availability
|
|
256
|
+
* @param {number} [now] epoch milliseconds; defaults to the current clock
|
|
257
|
+
* @returns {boolean}
|
|
258
|
+
*/
|
|
259
|
+
export function isRuntimeAvailable(availability, now = Date.now()) {
|
|
231
260
|
if (!availability) return false;
|
|
232
|
-
|
|
233
|
-
|
|
261
|
+
const rested = availability.available === true
|
|
262
|
+
? !availability.exhaustedUntil || Date.parse(availability.exhaustedUntil) <= now
|
|
263
|
+
: Boolean(availability.exhaustedUntil && Date.parse(availability.exhaustedUntil) <= now);
|
|
264
|
+
if (!rested) return false;
|
|
265
|
+
const windowSec = availability.window === undefined || availability.window === null
|
|
266
|
+
? undefined
|
|
267
|
+
: AVAILABILITY_WINDOW_SEC[availability.window];
|
|
268
|
+
if (windowSec === undefined || availability.observedAt === undefined || availability.observedAt === null) return true;
|
|
269
|
+
const observedAt = Date.parse(availability.observedAt);
|
|
270
|
+
return !Number.isNaN(observedAt) && observedAt + windowSec * 1000 >= now;
|
|
234
271
|
}
|
|
235
272
|
|
package/src/engine/scheduler.mjs
CHANGED
|
@@ -463,23 +463,23 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
463
463
|
// batched call it used to be, but the chain still runs them one at a time.
|
|
464
464
|
// A settlement may itself dispatch the node's next phase (a judge, a
|
|
465
465
|
// revision) through the same `startJudge`/`startWorker` calls dispatch below
|
|
466
|
-
// uses
|
|
467
|
-
//
|
|
468
|
-
//
|
|
469
|
-
//
|
|
470
|
-
//
|
|
471
|
-
//
|
|
466
|
+
// uses, so it is handed the real `running` to dispatch into: a job it starts
|
|
467
|
+
// is counted against `maxParallel` from the instant the process exists, and
|
|
468
|
+
// `applyRejection` reads that same map to decide whether the run has a slot
|
|
469
|
+
// for the revision at all. It used to dispatch into the throwaway one-entry
|
|
470
|
+
// map instead, copied back only once the settlement returned, which is how
|
|
471
|
+
// run state-location-and-routing-economics-13 came to hold two workers under
|
|
472
|
+
// `maxParallel: 1` on 2026-09-21. What it may *settle* is still only its own
|
|
473
|
+
// node: the one-entry map below is the job, not the run. The node's own
|
|
474
|
+
// steps stay ordered by never starting a second settlement for a node whose
|
|
475
|
+
// first has not yet cleared `pendingSettlements`.
|
|
472
476
|
const settleClosedJobsInBackground = () => {
|
|
473
477
|
for (const [nodeId, job] of [...running]) {
|
|
474
478
|
if (pendingSettlements.has(nodeId) || !job.closed || invocationAlive(job.invocation)) continue;
|
|
475
479
|
running.delete(nodeId);
|
|
476
|
-
const slot = new Map([[nodeId, job]]);
|
|
477
480
|
const settlement = settlementQueue
|
|
478
|
-
.then(() => finalizeClosedJobs(contract, runDir, states,
|
|
479
|
-
.finally(() =>
|
|
480
|
-
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
481
|
-
pendingSettlements.delete(nodeId);
|
|
482
|
-
});
|
|
481
|
+
.then(() => finalizeClosedJobs(contract, runDir, states, new Map([[nodeId, job]]), lock, campaign.path, running))
|
|
482
|
+
.finally(() => pendingSettlements.delete(nodeId));
|
|
483
483
|
// The queue itself must never reject -- a rejected settlement (a lost
|
|
484
484
|
// lock, a programmer error) would otherwise wedge every node queued
|
|
485
485
|
// behind it. The rejection still reaches whoever awaits the real
|
|
@@ -635,20 +635,16 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
635
635
|
// does, on the same per-run candidate ref and worktree
|
|
636
636
|
// `settlementQueue` exists to serialize -- so it is dispatched the
|
|
637
637
|
// same way: chained onto the queue rather than awaited here, using
|
|
638
|
-
// its
|
|
639
|
-
//
|
|
640
|
-
// a time, gated by `pendingSettlements`); only
|
|
641
|
-
// waiting behind it.
|
|
638
|
+
// `running` itself as its dispatch map, so the judge it starts is
|
|
639
|
+
// counted the instant it exists. The node's own order is untouched
|
|
640
|
+
// (still one entry at a time, gated by `pendingSettlements`); only
|
|
641
|
+
// the tick stops waiting behind it.
|
|
642
642
|
if (state.phase === "judge" && state.result) {
|
|
643
643
|
const workerResult = state.result;
|
|
644
|
-
const slot = new Map();
|
|
645
644
|
const settlement = settlementQueue
|
|
646
|
-
.then(() => startJudge(contract, node, state, runDir,
|
|
647
|
-
.then((round) => applyJudgeRound(round, contract, node, state, runDir,
|
|
648
|
-
.finally(() =>
|
|
649
|
-
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
650
|
-
pendingSettlements.delete(node.id);
|
|
651
|
-
});
|
|
645
|
+
.then(() => startJudge(contract, node, state, runDir, running, workerResult, lock, states, campaign.path))
|
|
646
|
+
.then((round) => applyJudgeRound(round, contract, node, state, runDir, running, lock, states, campaign.path, workerResult))
|
|
647
|
+
.finally(() => pendingSettlements.delete(node.id));
|
|
652
648
|
settlementQueue = settlement.catch(() => {});
|
|
653
649
|
settlement.catch((error) => {
|
|
654
650
|
if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
|
package/src/engine/scope.mjs
CHANGED
|
@@ -3,15 +3,19 @@
|
|
|
3
3
|
* touched, and what to do when those differ.
|
|
4
4
|
*
|
|
5
5
|
* Scope is advisory by design -- an unexpected write is recorded as a finding
|
|
6
|
-
* and shown to the judge, not treated as a crime -- with
|
|
6
|
+
* and shown to the judge, not treated as a crime -- with two exceptions:
|
|
7
7
|
* `resolveUnknownEffect` decides whether an invocation whose effect is unproven
|
|
8
|
-
* may be replayed at all, and a dirty scope there is a refusal
|
|
8
|
+
* may be replayed at all, and a dirty scope there is a refusal; and a write
|
|
9
|
+
* that lands on a file the node's own proof names is never deferred, because a
|
|
10
|
+
* verification that passes over an edited prover has proven nothing.
|
|
9
11
|
*/
|
|
12
|
+
import { relative, resolve } from "node:path";
|
|
13
|
+
|
|
10
14
|
import { SETTLED } from "./prompts.mjs";
|
|
11
15
|
import { appendTransitionEvent, recordExecutionOverride, transition, writeNode } from "./state.mjs";
|
|
12
16
|
import { attemptWorkspace } from "../repo/worktree.mjs";
|
|
13
17
|
|
|
14
|
-
import { errorCode, errorMessage, excerpt } from "../util.mjs";
|
|
18
|
+
import { errorCode, errorMessage, excerpt, isContained } from "../util.mjs";
|
|
15
19
|
import { executeControllerVerification } from "./verify.mjs";
|
|
16
20
|
import { providerReceiptsFromInvocationTail, settleInvocation } from "../run/operations.mjs";
|
|
17
21
|
import { readJson } from "../run/store.mjs";
|
|
@@ -98,6 +102,94 @@ export function workerScope(taskPacket) {
|
|
|
98
102
|
roots: taskPacket.writeRoots ?? [],
|
|
99
103
|
};
|
|
100
104
|
}
|
|
105
|
+
/**
|
|
106
|
+
* @typedef {{tokens: string[], cwd: string, literal: boolean, citation: string}} ProofCitation
|
|
107
|
+
*/
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Everything a node's own proofs name: a Definition of Done `path` proof's
|
|
111
|
+
* path, the words of a `command` proof (a command proof carries a display
|
|
112
|
+
* string, not an argv, so a quoted path holding a space is not recovered), the
|
|
113
|
+
* argv of the verification entry a `verification` proof references, and the
|
|
114
|
+
* argv of every verification command the packet declares.
|
|
115
|
+
*
|
|
116
|
+
* @param {ValidatedNode} node
|
|
117
|
+
* @returns {ProofCitation[]}
|
|
118
|
+
*/
|
|
119
|
+
function proofCitations(node) {
|
|
120
|
+
const commands = node.taskPacket.verification ?? [];
|
|
121
|
+
/** @type {ProofCitation[]} */
|
|
122
|
+
const citations = [];
|
|
123
|
+
// Definition of Done items come first so that a path both a checklist item
|
|
124
|
+
// and a verification command name is reported under the checklist item, the
|
|
125
|
+
// name a human reading the failure can act on.
|
|
126
|
+
for (const item of node.definitionOfDone ?? []) {
|
|
127
|
+
const proof = item.proof;
|
|
128
|
+
if (!proof) continue;
|
|
129
|
+
if (proof.kind === "path") {
|
|
130
|
+
citations.push({ tokens: [proof.ref], cwd: ".", literal: true, citation: `${item.id} path proof` });
|
|
131
|
+
} else if (proof.kind === "command") {
|
|
132
|
+
citations.push({ tokens: proof.ref.split(/\s+/u), cwd: ".", literal: false, citation: `${item.id} command proof` });
|
|
133
|
+
} else {
|
|
134
|
+
const command = commands[Number.parseInt(proof.ref, 10)];
|
|
135
|
+
if (command) citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `${item.id} verification[${proof.ref}] proof` });
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
for (const [index, command] of commands.entries()) {
|
|
139
|
+
citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `verification[${index}]` });
|
|
140
|
+
}
|
|
141
|
+
return citations;
|
|
142
|
+
}
|
|
143
|
+
const PATH_SEPARATOR = /[\\/]/u;
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* The unexpected writes that landed on a file the node's own proof names.
|
|
147
|
+
* Matching a command's argv against files is inherently approximate, so this is
|
|
148
|
+
* lexical and deliberately narrow. A word is read as a path only when it is not
|
|
149
|
+
* an option, is shaped like one (a `path` proof's ref, or a word carrying a
|
|
150
|
+
* separator or an extension), and resolves inside the workspace; it then claims
|
|
151
|
+
* an unexpected path it equals, or -- when it carries a separator or is a `path`
|
|
152
|
+
* proof's ref, so a directory really was named -- one it is the directory
|
|
153
|
+
* prefix of.
|
|
154
|
+
*
|
|
155
|
+
* What it deliberately does not catch: a file a proof reaches through a script
|
|
156
|
+
* (`npm test`), a shell string, or a glob the tool expands itself. The bare
|
|
157
|
+
* words of a command are never paths, which is what keeps the ordinary case
|
|
158
|
+
* advisory -- measured against the real node `requirement-ids-reach-the-node`
|
|
159
|
+
* (run `state-location-and-routing-economics-10-requirement-ids-and-closure`),
|
|
160
|
+
* whose legitimate out-of-scope write to `src/plan/freeze.mjs` is claimed by
|
|
161
|
+
* none of its proofs: not by `npm run typecheck`, not by the two test files its
|
|
162
|
+
* `command` proofs name, and not by the loose words of a quoted
|
|
163
|
+
* `--test-name-pattern`.
|
|
164
|
+
*
|
|
165
|
+
* @param {ValidatedNode} node
|
|
166
|
+
* @param {string[]} unexpectedPaths
|
|
167
|
+
* @param {string} workspace
|
|
168
|
+
* @returns {{path: string, citation: string}[]}
|
|
169
|
+
*/
|
|
170
|
+
function proofCitedWrites(node, unexpectedPaths, workspace) {
|
|
171
|
+
/** @type {Map<string, string>} */
|
|
172
|
+
const cited = new Map();
|
|
173
|
+
for (const citation of proofCitations(node)) {
|
|
174
|
+
const base = resolve(workspace, citation.cwd);
|
|
175
|
+
for (const token of citation.tokens) {
|
|
176
|
+
if (!token || token.startsWith("-")) continue;
|
|
177
|
+
const directory = citation.literal || PATH_SEPARATOR.test(token);
|
|
178
|
+
if (!directory && !/\.[A-Za-z0-9]+$/u.test(token)) continue;
|
|
179
|
+
const target = resolve(base, token);
|
|
180
|
+
if (!isContained(workspace, target)) continue;
|
|
181
|
+
const named = relative(workspace, target).replaceAll("\\", "/");
|
|
182
|
+
// The workspace root itself names no file in particular: a proof run from
|
|
183
|
+
// the root must not make every unexpected write a proof-citing one.
|
|
184
|
+
if (!named) continue;
|
|
185
|
+
for (const path of unexpectedPaths) {
|
|
186
|
+
if (cited.has(path)) continue;
|
|
187
|
+
if (path === named || (directory && path.startsWith(`${named}/`))) cited.set(path, citation.citation);
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
return [...cited].map(([path, citation]) => ({ path, citation }));
|
|
192
|
+
}
|
|
101
193
|
/**
|
|
102
194
|
* @param {ValidatedContract} contract
|
|
103
195
|
* @param {string} runDir
|
|
@@ -122,9 +214,19 @@ export function checkWorkerScope(contract, runDir, job, lock, options = {}) {
|
|
|
122
214
|
// A completed attempt whose controller verification passes never fails
|
|
123
215
|
// on scope alone (TECH-SPEC lean, rule 1): the caller defers the verdict
|
|
124
216
|
// until verification has run and records an advisory finding instead.
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
217
|
+
// The single exception is a write onto a file the node's own proof names:
|
|
218
|
+
// verification then passes because the attempt edited the thing doing the
|
|
219
|
+
// proving, and an advisory nobody must read before the gate is too weak a
|
|
220
|
+
// signal for that. Scanned over the bounded path list -- the same first 64
|
|
221
|
+
// paths every other surface reports.
|
|
222
|
+
const cited = proofCitedWrites(job.node, bounded.unexpectedPaths, job.cwd);
|
|
223
|
+
if (options.deferViolation && !cited.length) return true;
|
|
224
|
+
const message = cited.length
|
|
225
|
+
// Kept short on purpose: an error message is capped at 120 characters,
|
|
226
|
+
// and the proof that names the path is the part a reader cannot recover
|
|
227
|
+
// from `state.scope` afterwards.
|
|
228
|
+
? `proof-cited unexpected write (${cited.length}): ${cited.slice(0, 8).map(({ path, citation }) => `${path} (${citation})`).join(", ")}`
|
|
229
|
+
: `unexpected paths changed (${scope.unexpectedPaths.length}): ${bounded.unexpectedPaths.slice(0, 8).join(", ")}`;
|
|
128
230
|
if (!SETTLED.has(state.status)) {
|
|
129
231
|
transition(runDir, state, "failed", { phase: "worker", error: { code: "unexpected_write", message: excerpt(message) } }, lock);
|
|
130
232
|
appendTransitionEvent(runDir, state, "failed", "failed", {
|
package/src/engine/settle.mjs
CHANGED
|
@@ -69,18 +69,29 @@ export function applyRejection(contract, node, state, runDir, running, lock, sta
|
|
|
69
69
|
// dispatch below and a later scheduler dispatch (the `running` is null
|
|
70
70
|
// path) carry it.
|
|
71
71
|
state.previousAttempt = renderPreviousAttemptSection(state) ?? state.previousAttempt;
|
|
72
|
-
|
|
72
|
+
// A revision is a new attempt, and a new attempt is a dispatch: it starts
|
|
73
|
+
// here only against a slot the run actually has free. `running` is the
|
|
74
|
+
// scheduler's own map, with this node's own closed job already out of it,
|
|
75
|
+
// so `running.size` is exactly what the dispatch loop's own
|
|
76
|
+
// `maxParallel - running.size` will read. A settlement that dispatched
|
|
77
|
+
// regardless is how run state-location-and-routing-economics-13 ran two
|
|
78
|
+
// workers under `maxParallel: 1` on 2026-09-21, and two of that day's
|
|
79
|
+
// three OOM kills happened with more running than the contract declared.
|
|
80
|
+
if (running && running.size < contract.maxParallel) {
|
|
73
81
|
// Dispatching here owns the increment, because `startWorker` expects the
|
|
74
82
|
// attempt number it is about to run under.
|
|
75
83
|
state.attempt += 1;
|
|
76
84
|
startWorker(contract, node, state, runDir, running, retryPrompt(node, verdict), lock, states, campaignPath, { forceFresh });
|
|
77
85
|
return;
|
|
78
86
|
}
|
|
79
|
-
// Handing the node back to the scheduler instead
|
|
80
|
-
//
|
|
81
|
-
//
|
|
82
|
-
//
|
|
83
|
-
//
|
|
87
|
+
// Handing the node back to the scheduler instead -- because there is no
|
|
88
|
+
// loop to hand it to (recovery, resume), or because the run is full and
|
|
89
|
+
// the scheduler is the one place that knows when it stops being full. Its
|
|
90
|
+
// dispatch increments on the way out, so incrementing here too spent two
|
|
91
|
+
// attempt numbers on one retry. Observed 2026-09-13 on a resume after a
|
|
92
|
+
// killed controller — a node that ran twice reported attempt 3, with no
|
|
93
|
+
// `…2.*` logs and a `worktree.previousAttempt` naming an attempt that
|
|
94
|
+
// never existed.
|
|
84
95
|
transition(runDir, state, "pending", { phase: "worker", error: null }, lock);
|
|
85
96
|
return;
|
|
86
97
|
}
|
|
@@ -91,7 +102,7 @@ export function applyRejection(contract, node, state, runDir, running, lock, sta
|
|
|
91
102
|
}, lock);
|
|
92
103
|
}
|
|
93
104
|
/** Deterministic verification failure settles through the shared rejection path. The verdict carries this attempt's unexpected paths, so a red attempt reports them whether it stops here or starts its revision (TECH-SPEC lean, rule 1). @param {ValidatedContract} contract @param {ValidatedNode} node @param {NodeSnapshot} state @param {string} runDir @param {Map<string, Job>|null} running @param {LockHandle} lock @param {Map<string, NodeSnapshot>} states @param {string} campaignPath @param {JudgeVerdict} [verdict] */
|
|
94
|
-
export function applyVerificationFailure(contract, node, state, runDir, running, lock, states, campaignPath, verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope)) {
|
|
105
|
+
export function applyVerificationFailure(contract, node, state, runDir, running, lock, states, campaignPath, verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope)) {
|
|
95
106
|
applyRejection(contract, node, state, runDir, running, lock, states, campaignPath, verdict, { code: "verification_failed", label: "verification" });
|
|
96
107
|
}
|
|
97
108
|
|
|
@@ -193,7 +204,7 @@ export async function settleDone(contract, node, state, runDir, lock, states, ca
|
|
|
193
204
|
removeWorktree(contract.cwd, acceptedPath);
|
|
194
205
|
},
|
|
195
206
|
onVerificationFailure: async (transaction) => {
|
|
196
|
-
const verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope);
|
|
207
|
+
const verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope);
|
|
197
208
|
verdict.summary = "integrated candidate verification failed";
|
|
198
209
|
const divergent = candidateOnlyFailures(state.verification, transaction.candidateEvidence);
|
|
199
210
|
verdict.findings = [...(verdict.findings ?? []), {
|
package/src/host/preflight.mjs
CHANGED
|
@@ -27,6 +27,7 @@ import { errorMessage } from "../util.mjs";
|
|
|
27
27
|
import { boundedGitSync } from "../repo/worktree.mjs";
|
|
28
28
|
import { routeRuntime } from "../contract/runtime.mjs";
|
|
29
29
|
import { NOTIFY_BIN_ENV, noTransportWarning } from "../notify/index.mjs";
|
|
30
|
+
import { NOTIFY_SESSION_ENV, sessionWakeNotice } from "../notify/session.mjs";
|
|
30
31
|
import { findExecutable } from "./platform.mjs";
|
|
31
32
|
import { colorLevel, statusToken } from "../cli/brand.mjs";
|
|
32
33
|
import { RUNS_DIR_NAME } from "../run/paths.mjs";
|
|
@@ -209,9 +210,9 @@ export function environmentPreflight(options) {
|
|
|
209
210
|
*/
|
|
210
211
|
export function notifyTransportCheck(env = process.env) {
|
|
211
212
|
const warning = noTransportWarning(env);
|
|
212
|
-
return warning
|
|
213
|
-
|
|
214
|
-
|
|
213
|
+
if (warning) return fail("notify transport", warning, true);
|
|
214
|
+
const external = env[NOTIFY_BIN_ENV] ? `${NOTIFY_BIN_ENV}=${env[NOTIFY_BIN_ENV]}` : `${NOTIFY_BIN_ENV} unset`;
|
|
215
|
+
return pass("notify transport", `${external} · ${sessionWakeNotice(env)}`);
|
|
215
216
|
}
|
|
216
217
|
|
|
217
218
|
/** @param {EnvReport} report @returns {EnvCheck[]} the checks that block a dispatch */
|
|
@@ -265,6 +266,7 @@ export function declaredVerificationCommands(contract) {
|
|
|
265
266
|
*/
|
|
266
267
|
const SIDE_EFFECT_ENV_KEYS = [
|
|
267
268
|
NOTIFY_BIN_ENV, // a measurement must not notify a human
|
|
269
|
+
NOTIFY_SESSION_ENV, // nor wake the harness session it was measured from
|
|
268
270
|
"FABERUN_CODEX_BIN", // could redirect the timed command at a live, paid codex binary instead of this repository's own fixtures
|
|
269
271
|
"FABERUN_CLAUDE_BIN", // same, for the claude harness
|
|
270
272
|
"FABERUN_AGY_BIN", // same, for the agy harness
|