faberun 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/campaign/index.mjs +74 -2
- package/src/campaign/record.mjs +28 -0
- package/src/cli/plan.mjs +40 -1
- package/src/cli.mjs +1 -0
- package/src/contract/index.mjs +15 -4
- package/src/contract/runtime.mjs +50 -2
- package/src/contract/snapshot.mjs +21 -2
- package/src/contract/verification.mjs +26 -10
- package/src/engine/assignment.mjs +33 -6
- package/src/engine/backoff.mjs +11 -5
- package/src/engine/failover.mjs +62 -1
- package/src/engine/judge-gate.mjs +63 -5
- package/src/engine/lifecycle.mjs +17 -4
- package/src/engine/mutation.mjs +47 -7
- package/src/engine/resume.mjs +1 -1
- package/src/engine/runtime-discovery.mjs +44 -7
- package/src/engine/scheduler.mjs +19 -23
- package/src/engine/scope.mjs +108 -6
- package/src/engine/settle.mjs +22 -8
- package/src/engine/state.mjs +1 -0
- package/src/plan/freeze.mjs +29 -0
- package/src/plan/pipeline.mjs +203 -22
- package/src/plan/routing.mjs +151 -33
- package/src/plan/sizing.mjs +45 -0
- package/src/report/final.mjs +2 -1
- package/src/report/packet-repetition.mjs +121 -0
- package/src/report/render.mjs +3 -3
package/src/engine/failover.mjs
CHANGED
|
@@ -8,9 +8,14 @@
|
|
|
8
8
|
* runtime a role started on, and (if declared) the runtime its `fallback`
|
|
9
9
|
* names. contract.mjs validates the field is never a self-loop; because a
|
|
10
10
|
* hop is bounded at one, a multi-runtime cycle is structurally impossible.
|
|
11
|
+
*
|
|
12
|
+
* Per-attempt resolution (`routeRuntimeForState`) lives here too, because
|
|
13
|
+
* attempt affinity is read against the same failover facts: the runtime the
|
|
14
|
+
* previous attempt ran is preferred until the hop, the catalogue, or the
|
|
15
|
+
* judge's vendor rule disqualifies it.
|
|
11
16
|
*/
|
|
12
17
|
import { harnessCapabilities } from "../harnesses/index.mjs";
|
|
13
|
-
import { nextSameTierRuntime } from "./runtime-discovery.mjs";
|
|
18
|
+
import { isRuntimeAvailable, nextSameTierRuntime } from "./runtime-discovery.mjs";
|
|
14
19
|
import { routeRuntime } from "../contract/runtime.mjs";
|
|
15
20
|
|
|
16
21
|
/** @typedef {import("../contract/index.mjs").ValidatedNode} ValidatedNode */
|
|
@@ -170,6 +175,50 @@ export function routingBackoffActive(state, phase) {
|
|
|
170
175
|
return Boolean(override?.role === phase && override.backoffUntil && Date.parse(override.backoffUntil) > Date.now());
|
|
171
176
|
}
|
|
172
177
|
|
|
178
|
+
/**
|
|
179
|
+
* The runtime id this role's previous attempt on this node ran on, read off
|
|
180
|
+
* the durable invocation record. The invocations are the one record that
|
|
181
|
+
* survives both a routed hop and the revision boundary: `resetPhaseRouting`
|
|
182
|
+
* clears the routing override between revisions, but never the invocations.
|
|
183
|
+
*
|
|
184
|
+
* @param {NodeSnapshot} state
|
|
185
|
+
* @param {"worker"|"judge"} role
|
|
186
|
+
* @returns {string|undefined}
|
|
187
|
+
*/
|
|
188
|
+
function previousAttemptRuntimeId(state, role) {
|
|
189
|
+
const found = [...(state.invocations ?? [])].reverse().find((invocation) => invocation.phase === role)?.runtimeId;
|
|
190
|
+
return typeof found === "string" ? found : undefined;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* May attempt affinity keep this role on `candidate` -- the runtime its
|
|
195
|
+
* previous attempt on this node ran? Affinity yields to a known exhaustion
|
|
196
|
+
* only: a catalogue record that `isRuntimeAvailable` -- the one home of the
|
|
197
|
+
* exhaustion and staleness rule -- refuses to admit. An absent record is not
|
|
198
|
+
* that: the snapshot's catalogue copy is the classified fields at best and
|
|
199
|
+
* often empty outright, and the strongest evidence of health here is the
|
|
200
|
+
* attempt that just ran on this runtime, so absence of a record must not read
|
|
201
|
+
* as exhaustion any more than R12 lets it read as rested. A judge candidate
|
|
202
|
+
* stays bound by the cross-vendor rule its assignment was composed under,
|
|
203
|
+
* compared against the worker that actually ran the node, exactly as
|
|
204
|
+
* `nextSameTierRuntime` compares its own candidates.
|
|
205
|
+
*
|
|
206
|
+
* @param {ValidatedContract} contract
|
|
207
|
+
* @param {NodeSnapshot} state
|
|
208
|
+
* @param {"worker"|"judge"} role
|
|
209
|
+
* @param {string} candidate
|
|
210
|
+
* @returns {boolean}
|
|
211
|
+
*/
|
|
212
|
+
function admitsAffinity(contract, state, role, candidate) {
|
|
213
|
+
const runtime = contract.runtimes[candidate];
|
|
214
|
+
if (!runtime) return false;
|
|
215
|
+
const availability = state.routing?.availability?.[candidate];
|
|
216
|
+
if (availability && !isRuntimeAvailable(availability)) return false;
|
|
217
|
+
if (role !== "judge") return true;
|
|
218
|
+
const workerId = previousAttemptRuntimeId(state, "worker") ?? state.routing?.assignments?.worker;
|
|
219
|
+
return runtime.vendor !== (workerId ? contract.runtimes[workerId]?.vendor : null);
|
|
220
|
+
}
|
|
221
|
+
|
|
173
222
|
/**
|
|
174
223
|
* @param {ValidatedContract} contract
|
|
175
224
|
* @param {ValidatedNode} node
|
|
@@ -183,6 +232,18 @@ export function routeRuntimeForState(contract, node, state, role) {
|
|
|
183
232
|
const runtime = contract.runtimes[override.runtime];
|
|
184
233
|
return { id: override.runtime, ...runtime, capabilities: harnessCapabilities(runtime) };
|
|
185
234
|
}
|
|
235
|
+
// Attempt affinity: successive attempts and revisions of this node prefer
|
|
236
|
+
// the runtime the previous attempt ran, because it is the one holding the
|
|
237
|
+
// node's context. It ranks ahead of the frozen assignment -- the assignment
|
|
238
|
+
// decided the first attempt, the attempt that ran since decides the next --
|
|
239
|
+
// and below the role-matched override above, which is itself already an
|
|
240
|
+
// affinity outcome: a reset hold on the warm runtime, or the failover edge
|
|
241
|
+
// affinity yielded to.
|
|
242
|
+
const previous = previousAttemptRuntimeId(state, role);
|
|
243
|
+
if (previous !== undefined && admitsAffinity(contract, state, role, previous)) {
|
|
244
|
+
const runtime = contract.runtimes[previous];
|
|
245
|
+
return { id: previous, ...runtime, capabilities: harnessCapabilities(runtime) };
|
|
246
|
+
}
|
|
186
247
|
const assigned = state.routing?.assignments?.[role];
|
|
187
248
|
if (assigned && contract.runtimes[assigned]) {
|
|
188
249
|
const runtime = contract.runtimes[assigned];
|
|
@@ -14,8 +14,10 @@ import { stat } from "node:fs/promises";
|
|
|
14
14
|
import { isAbsolute, resolve } from "node:path";
|
|
15
15
|
import { reviewMode, UNCITED_REJECTION_REASON } from "../contract/review-modes.mjs";
|
|
16
16
|
import { JUDGE_LIMITS } from "../contract/judge-envelope.mjs";
|
|
17
|
+
import { sharedVerificationCommands } from "../contract/final-verification.mjs";
|
|
17
18
|
|
|
18
19
|
/** @typedef {import("../contract/definition-of-done.mjs").DefinitionOfDoneItem} DefinitionOfDoneItem */
|
|
20
|
+
/** @typedef {import("../contract/verification.mjs").VerificationCommand} VerificationCommand */
|
|
19
21
|
/** @typedef {import("../contract/definition-of-done.mjs").DefinitionOfDoneProof} DefinitionOfDoneProof */
|
|
20
22
|
/** @typedef {import("../contract/index.mjs").ExecutionOverride} ExecutionOverride */
|
|
21
23
|
/** @typedef {import("../contract/index.mjs").NodeSnapshot} NodeSnapshot */
|
|
@@ -349,6 +351,48 @@ function declaredWriteCoverage(state) {
|
|
|
349
351
|
directoryRoots.some((root) => path === root || path.startsWith(`${root}/`));
|
|
350
352
|
}
|
|
351
353
|
|
|
354
|
+
/**
|
|
355
|
+
* Whether a path named by verification output belongs to the contract's own
|
|
356
|
+
* `sharedVerification` suite -- the repository ratchets the operator appends to
|
|
357
|
+
* every node -- rather than to the node's own work. An argv entry names either
|
|
358
|
+
* the file the runner was given or a directory it walked.
|
|
359
|
+
*
|
|
360
|
+
* Deliberately not caught: a `sharedVerification` that reaches the ratchet
|
|
361
|
+
* without naming it in argv (`npm test`, or any script that picks the files
|
|
362
|
+
* itself). No token matches, the file is not recognized as a ratchet, and the
|
|
363
|
+
* contract-defect advice applies to it as it did before. Widening the match to
|
|
364
|
+
* guess what a script runs would be worse than the gap it closes.
|
|
365
|
+
*
|
|
366
|
+
* @param {{sharedVerification?: VerificationCommand[]}} contract
|
|
367
|
+
* @returns {(path: string) => boolean}
|
|
368
|
+
*/
|
|
369
|
+
function sharedVerificationCoverage(contract) {
|
|
370
|
+
const argv = sharedVerificationCommands(contract).flatMap((command) => command.argv);
|
|
371
|
+
return (path) => argv.some((token) => token === path || path.startsWith(`${token}/`));
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* The operator-facing description for a failing ratchet: a check the contract
|
|
376
|
+
* itself declared, that runs on every node, and that this node's change broke.
|
|
377
|
+
*
|
|
378
|
+
* It says nothing about the write scope on purpose. Telling the operator to
|
|
379
|
+
* hand the node the ratchet licenses the next worker to edit the rule it just
|
|
380
|
+
* violated -- observed 2026-09-21, when a node that pushed
|
|
381
|
+
* `src/report/render.mjs` to 801 lines was advised to add the 800-line ratchet
|
|
382
|
+
* to its `writeFiles`. The remedy is already inside the node's scope: its own
|
|
383
|
+
* code.
|
|
384
|
+
*
|
|
385
|
+
* @param {string[]} paths
|
|
386
|
+
* @returns {string}
|
|
387
|
+
*/
|
|
388
|
+
function ratchetFailureDescription(paths) {
|
|
389
|
+
const files = paths.join(", ");
|
|
390
|
+
return boundedText(
|
|
391
|
+
`deterministic verification failed in ${files}, which the contract runs on every node as sharedVerification: this node's change broke a repository-wide rule. The remedy is in the node's own code -- bring the change back within the rule the check enforces. The check itself is not the node's to change, and relaxing it is not a fix`,
|
|
392
|
+
JUDGE_LIMITS.descriptionBytes,
|
|
393
|
+
);
|
|
394
|
+
}
|
|
395
|
+
|
|
352
396
|
/**
|
|
353
397
|
* The operator-facing description for a failure that named a test outside the
|
|
354
398
|
* declared write scope. The defect is not in the worker's code: the worker is
|
|
@@ -372,24 +416,38 @@ function undeclaredTestDescription(paths) {
|
|
|
372
416
|
* The deterministic controller-verification failure verdict, kept next to the
|
|
373
417
|
* Definition of Done gate so every deterministic failure settles identically.
|
|
374
418
|
*
|
|
419
|
+
* The contract is a parameter because an undeclared test file has two opposite
|
|
420
|
+
* remedies and only the contract tells them apart: a ratchet it declared in
|
|
421
|
+
* `sharedVerification` is the node's code to fix, while any other withheld test
|
|
422
|
+
* is the contract's scope to widen.
|
|
423
|
+
*
|
|
424
|
+
* @param {{sharedVerification?: VerificationCommand[]}} contract
|
|
375
425
|
* @param {{verification?: {commands?: Array<{argv: string[], passed?: boolean, attempts?: Array<{stdout?: string, stderr?: string, exitCode?: number|null, timedOut?: boolean}>}>, error?: unknown}|null, scope?: {boundary?: {files?: string[], roots?: string[], fileRoots?: string[]}}|null}} state
|
|
376
426
|
* @returns {import("./prompts.mjs").JudgeVerdict}
|
|
377
427
|
*/
|
|
378
|
-
export function verificationFailureVerdict(state) {
|
|
428
|
+
export function verificationFailureVerdict(contract, state) {
|
|
379
429
|
const failedCommands = (state.verification?.commands ?? []).filter((command) => !command.passed);
|
|
380
430
|
const evidence = failedCommands.length
|
|
381
431
|
? failedCommands.map((command) => `${command.argv.join(" ")}: ${(command.attempts ?? []).map((attempt) => `exit=${attempt.exitCode ?? "-"}${attempt.timedOut ? " timeout" : ""}`).join(", ")}`).join("; ")
|
|
382
432
|
: state.verification?.error ?? "verification controller failed to execute a command";
|
|
383
|
-
const
|
|
433
|
+
const declared = declaredWriteCoverage(state);
|
|
434
|
+
const undeclared = namedTestFiles(failedCommands).filter((path) => !declared(path));
|
|
435
|
+
const isRatchet = sharedVerificationCoverage(contract);
|
|
436
|
+
const ratchets = undeclared.filter(isRatchet);
|
|
437
|
+
const withheld = undeclared.filter((path) => !isRatchet(path));
|
|
438
|
+
const descriptions = [
|
|
439
|
+
...(ratchets.length ? [ratchetFailureDescription(ratchets)] : []),
|
|
440
|
+
...(withheld.length ? [undeclaredTestDescription(withheld)] : []),
|
|
441
|
+
];
|
|
384
442
|
return {
|
|
385
443
|
verdict: "fail",
|
|
386
444
|
maxSeverity: "critical",
|
|
387
445
|
summary: "deterministic verification failed",
|
|
388
|
-
findings: [{
|
|
446
|
+
findings: (descriptions.length ? descriptions : ["deterministic verification failed"]).map((description) => ({
|
|
389
447
|
severity: "critical",
|
|
390
|
-
description
|
|
448
|
+
description,
|
|
391
449
|
evidence: boundedText(evidence),
|
|
392
|
-
}
|
|
450
|
+
})),
|
|
393
451
|
};
|
|
394
452
|
}
|
|
395
453
|
|
package/src/engine/lifecycle.mjs
CHANGED
|
@@ -280,22 +280,35 @@ export function clearTierExhaustion(state) {
|
|
|
280
280
|
}
|
|
281
281
|
|
|
282
282
|
/**
|
|
283
|
+
* Settle what a closed invocation produced, and start whatever the outcome
|
|
284
|
+
* earns next.
|
|
285
|
+
*
|
|
286
|
+
* The two maps are two different things, and conflating them is what let a run
|
|
287
|
+
* exceed its own `maxParallel` (measured 2026-09-21, run
|
|
288
|
+
* state-location-and-routing-economics-13): `closed` is the job this call owns
|
|
289
|
+
* -- the scheduler hands one node's job at a time so two settlements never
|
|
290
|
+
* interleave -- while `running` is the run's live dispatch authority, the very
|
|
291
|
+
* map the scheduler's `maxParallel - running.size` counts. Anything started
|
|
292
|
+
* here lands in `running` and is accounted from that instant; nothing started
|
|
293
|
+
* here may be settled by this call.
|
|
294
|
+
*
|
|
283
295
|
* @param {ValidatedContract} contract
|
|
284
296
|
* @param {string} runDir
|
|
285
297
|
* @param {Map<string, NodeSnapshot>} states
|
|
286
|
-
* @param {Map<string, Job>}
|
|
298
|
+
* @param {Map<string, Job>} closed
|
|
287
299
|
* @param {LockHandle} lock
|
|
288
300
|
* @param {string} campaignPath
|
|
301
|
+
* @param {Map<string, Job>} running
|
|
289
302
|
* @returns {Promise<void>}
|
|
290
303
|
*/
|
|
291
|
-
export async function finalizeClosedJobs(contract, runDir, states,
|
|
304
|
+
export async function finalizeClosedJobs(contract, runDir, states, closed, lock, campaignPath, running) {
|
|
292
305
|
// Advisory spend lines are checked every tick, before outcome handling: a
|
|
293
306
|
// crossing must be visible while the spend is happening, not only when the
|
|
294
307
|
// run is already over. The check never stops or transitions a node.
|
|
295
308
|
await emitNodeAdvisories(contract, runDir, states);
|
|
296
|
-
for (const [nodeId, job] of
|
|
309
|
+
for (const [nodeId, job] of closed) {
|
|
297
310
|
if (!job.closed || invocationAlive(job.invocation)) continue;
|
|
298
|
-
|
|
311
|
+
closed.delete(nodeId);
|
|
299
312
|
const state = states.get(nodeId);
|
|
300
313
|
if (!state) continue;
|
|
301
314
|
// Usage is extracted and persisted BEFORE any outcome-specific handling:
|
package/src/engine/mutation.mjs
CHANGED
|
@@ -9,6 +9,14 @@
|
|
|
9
9
|
* break syntax and the sample is a deterministic eight, because a gate that
|
|
10
10
|
* fails at random, or on a mutant that does not compile, is worse than no gate.
|
|
11
11
|
*
|
|
12
|
+
* The kill fraction is declared by tier, not hand-picked per entry: the entry
|
|
13
|
+
* declares a risk tier and `MUTATION_TIERS` (`contract/verification.mjs`)
|
|
14
|
+
* fixes what each tier demands, so two nodes at the same tier sit the same
|
|
15
|
+
* bar. The whole entry — baseline plus sample — is held inside
|
|
16
|
+
* `MUTATION_TIME_BUDGET_MS`, and mutants the budget cannot afford stay in the
|
|
17
|
+
* denominator, so a suite too slow to sample inside the budget fails rather
|
|
18
|
+
* than passing on a narrowed set.
|
|
19
|
+
*
|
|
12
20
|
* It is a separate module from `run-command.mjs` so that the "doing" of a
|
|
13
21
|
* verification run and the "which file to break" policy do not grow together;
|
|
14
22
|
* the runner receives the argv executor as a callback rather than importing it,
|
|
@@ -16,10 +24,27 @@
|
|
|
16
24
|
*/
|
|
17
25
|
import { existsSync, readFileSync, statSync, writeFileSync } from "node:fs";
|
|
18
26
|
import { resolve } from "node:path";
|
|
27
|
+
import { MUTATION_TIERS } from "../contract/verification.mjs";
|
|
19
28
|
|
|
20
29
|
/** At most this many mutants run for one verification entry. */
|
|
21
30
|
export const MUTATION_BUDGET = 8;
|
|
22
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Wall-clock budget for one mutation entry — baseline plus every mutant run —
|
|
34
|
+
* in milliseconds. Each mutant re-runs the same argv, so each one costs what
|
|
35
|
+
* the baseline cost, and the runner refuses to start a mutant the remaining
|
|
36
|
+
* budget cannot afford.
|
|
37
|
+
*
|
|
38
|
+
* measured 2026-09-21 on this repository: the scenarios in
|
|
39
|
+
* test/engine/mutation.test.mjs cost 0.08–0.37s per run, so a node-sized entry
|
|
40
|
+
* (baseline plus eight mutants over a targeted suite) costs ~1–4s and samples
|
|
41
|
+
* in full. The budget admits any suite up to ~3.3s per run and denies slower
|
|
42
|
+
* ones every mutant — measured counter-example: `node --test test/contract`
|
|
43
|
+
* costs ~20s per run, and a mutation entry is scoped to the node's own
|
|
44
|
+
* verification command, not to a whole suite directory.
|
|
45
|
+
*/
|
|
46
|
+
export const MUTATION_TIME_BUDGET_MS = 30_000;
|
|
47
|
+
|
|
23
48
|
/**
|
|
24
49
|
* The six operators, and only these: comparison and logical swaps. Each swap
|
|
25
50
|
* keeps the expression well formed, so a surviving mutant is signal about the
|
|
@@ -109,18 +134,31 @@ function evenlySpaced(candidates, budget) {
|
|
|
109
134
|
*/
|
|
110
135
|
export async function runMutation(command, baseCwd, options) {
|
|
111
136
|
const mutation = command.mutation;
|
|
112
|
-
if (!mutation) throw new TypeError("mutation runner requires a command.mutation
|
|
113
|
-
const threshold = mutation.
|
|
114
|
-
const
|
|
137
|
+
if (!mutation) throw new TypeError("mutation runner requires a command.mutation risk tier");
|
|
138
|
+
const threshold = MUTATION_TIERS[mutation.tier];
|
|
139
|
+
const sample = evenlySpaced(mutationCandidates(baseCwd, options.writeFiles ?? []), MUTATION_BUDGET);
|
|
115
140
|
/** @type {VerificationAttemptResult[]} */
|
|
116
141
|
const attempts = [];
|
|
117
142
|
const baseline = await options.run(1);
|
|
118
143
|
attempts.push(baseline);
|
|
119
|
-
if (!baseline.passed) return { passed: false, killed: 0, total:
|
|
144
|
+
if (!baseline.passed) return { passed: false, killed: 0, total: sample.length, threshold, attempts };
|
|
145
|
+
// Every mutant re-runs the same argv, so the baseline's own measured duration
|
|
146
|
+
// is what one more attempt costs, and the budget is spent in those units. An
|
|
147
|
+
// executor that reports no duration (the unit-test seam) is charged zero and
|
|
148
|
+
// is never denied a mutant.
|
|
149
|
+
const attemptMs = Number.isFinite(baseline.durationMs) ? /** @type {number} */ (baseline.durationMs) : 0;
|
|
120
150
|
/** @type {Map<string, string>} */
|
|
121
151
|
const originals = new Map();
|
|
122
152
|
let killed = 0;
|
|
123
|
-
|
|
153
|
+
let spentMs = attemptMs;
|
|
154
|
+
let attempt = 1;
|
|
155
|
+
for (const candidate of sample) {
|
|
156
|
+
// Mutants the budget cannot afford are not silently dropped from the
|
|
157
|
+
// denominator: `total` stays the full sample, so truncation can only lower
|
|
158
|
+
// the kill fraction. A suite too slow to sample inside the budget fails a
|
|
159
|
+
// gate it is too slow to sit for — the narrowing the budget exists to
|
|
160
|
+
// prevent.
|
|
161
|
+
if (attemptMs > 0 && spentMs + attemptMs > MUTATION_TIME_BUDGET_MS) break;
|
|
124
162
|
let original = originals.get(candidate.absolute);
|
|
125
163
|
if (original === undefined) {
|
|
126
164
|
original = readFileSync(candidate.absolute, "utf8");
|
|
@@ -129,16 +167,18 @@ export async function runMutation(command, baseCwd, options) {
|
|
|
129
167
|
const mutated = `${original.slice(0, candidate.offset)}${candidate.to}${original.slice(candidate.offset + candidate.from.length)}`;
|
|
130
168
|
/** @type {VerificationAttemptResult} */
|
|
131
169
|
let result;
|
|
170
|
+
attempt += 1;
|
|
132
171
|
try {
|
|
133
172
|
writeFileSync(candidate.absolute, mutated);
|
|
134
|
-
result = await options.run(
|
|
173
|
+
result = await options.run(attempt);
|
|
135
174
|
} finally {
|
|
136
175
|
writeFileSync(candidate.absolute, original);
|
|
137
176
|
}
|
|
138
177
|
attempts.push(result);
|
|
178
|
+
spentMs += Number.isFinite(result.durationMs) ? /** @type {number} */ (result.durationMs) : 0;
|
|
139
179
|
if (!result.passed) killed += 1;
|
|
140
180
|
}
|
|
141
|
-
const total =
|
|
181
|
+
const total = sample.length;
|
|
142
182
|
// No operators to break is a vacuous pass: there is no mutant the suite could
|
|
143
183
|
// have failed to kill. The caller declared the entry, so an empty target is a
|
|
144
184
|
// measurement of nothing, not a suite that proved nothing.
|
package/src/engine/resume.mjs
CHANGED
|
@@ -609,7 +609,7 @@ export async function recoverIntegrationTransactions(contract, runDir, states, l
|
|
|
609
609
|
const node = contract.nodes.find((candidate) => candidate.id === transaction.node);
|
|
610
610
|
const state = states.get(transaction.node);
|
|
611
611
|
if (!node || !state || SETTLED.has(state.status)) return;
|
|
612
|
-
const verdict = verificationFailureWithScope(verificationFailureVerdict(state), state.scope);
|
|
612
|
+
const verdict = verificationFailureWithScope(verificationFailureVerdict(contract, state), state.scope);
|
|
613
613
|
verdict.summary = "integrated candidate verification failed during recovery";
|
|
614
614
|
applyRejection(contract, node, state, runDir, null, lock, states, campaignPath, verdict, {
|
|
615
615
|
code: "verification_failed",
|
|
@@ -9,7 +9,14 @@ export { exhaustedUntilOf, normalizeProviderAvailability } from "../harnesses/in
|
|
|
9
9
|
/** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
|
|
10
10
|
/** @typedef {{harness?: string, model?: string, vendor: string, tier?: number|string, costRank?: number, [key: string]: unknown}} RuntimeLike */
|
|
11
11
|
/** @typedef {{runtimes: Record<string, RuntimeLike>, runtimeDefaults?: {worker?: string, judge?: string}, nodes?: {id: string, runtime?: string, gate: {enabled: boolean, runtime?: string}}[]}} RuntimeContract */
|
|
12
|
-
/**
|
|
12
|
+
/**
|
|
13
|
+
* One runtime's catalogue record: what the harness de facto reported, and
|
|
14
|
+
* when. An unobservable datum is null -- never zero and never full allowance,
|
|
15
|
+
* so a runtime that reports nothing cannot look rested -- and an absent key on
|
|
16
|
+
* a record that predates the field reads as null at every reader. An
|
|
17
|
+
* observation older than its own window reads as unknown (`isRuntimeAvailable`).
|
|
18
|
+
* @typedef {{available: boolean, exhaustedUntil: string|null, reason: string, observedAt?: string|null, window?: string|null, remaining?: number|null}} RuntimeAvailability
|
|
19
|
+
*/
|
|
13
20
|
/** @typedef {{harness: string, model: string, vendor: string, tier: number, costRank: number, config?: Record<string, unknown>}} DiscoveryRuntime */
|
|
14
21
|
/** @typedef {{id: string, runtime: RuntimeLike, order: number}} RuntimeCandidate */
|
|
15
22
|
/** @typedef {import("../host/config.mjs").UserConfig} UserConfig */
|
|
@@ -89,7 +96,7 @@ export async function discoverRuntimes(runtimes, options = {}) {
|
|
|
89
96
|
*/
|
|
90
97
|
export function availableCandidates(runtimes, availability = {}) {
|
|
91
98
|
return Object.entries(runtimes)
|
|
92
|
-
.filter(([id]) =>
|
|
99
|
+
.filter(([id]) => isRuntimeAvailable(availability[id]))
|
|
93
100
|
.map(([id, runtime], order) => ({ id, runtime, order }));
|
|
94
101
|
}
|
|
95
102
|
|
|
@@ -172,7 +179,7 @@ export function nextSameTierRuntime(contract, stateRouting, role, current, attem
|
|
|
172
179
|
const used = new Set(attempted);
|
|
173
180
|
return Object.entries(contract.runtimes)
|
|
174
181
|
.filter(([id, runtime]) => id !== current && !used.has(id) && sameTier(runtime, currentRuntime))
|
|
175
|
-
.filter(([id]) =>
|
|
182
|
+
.filter(([id]) => isRuntimeAvailable(stateRouting.availability?.[id]))
|
|
176
183
|
.filter(([, runtime]) => role !== "judge" || runtime.vendor !== workerVendor)
|
|
177
184
|
.sort((left, right) => runtimeOrder(left[1]) - runtimeOrder(right[1]))
|
|
178
185
|
.map(([id]) => id)
|
|
@@ -226,10 +233,40 @@ function tierOrder(runtime) {
|
|
|
226
233
|
return typeof runtime.tier === "number" ? runtime.tier : runtime.costRank ?? Number.MAX_SAFE_INTEGER;
|
|
227
234
|
}
|
|
228
235
|
|
|
229
|
-
/**
|
|
230
|
-
|
|
236
|
+
/**
|
|
237
|
+
* Span in seconds of every rate-limit window label a harness reports. The
|
|
238
|
+
* labels are claude's `rateLimitType` values (measured 2026-09-17, the
|
|
239
|
+
* `rate_limit_event` line recorded in `src/harnesses/protocol.mjs`); a label
|
|
240
|
+
* missing here cannot prove staleness, so its observation never self-expires.
|
|
241
|
+
*
|
|
242
|
+
* @type {Readonly<Record<string, number>>}
|
|
243
|
+
*/
|
|
244
|
+
const AVAILABILITY_WINDOW_SEC = Object.freeze({ five_hour: 5 * 3600, seven_day: 7 * 86400 });
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* May a runtime be admitted on this catalogue record? Exhaustion is waited
|
|
248
|
+
* out on `exhaustedUntil`; an observation older than its own window reads as
|
|
249
|
+
* unknown and admits nothing, because unknown must not look rested. This is
|
|
250
|
+
* the one home of the rule: plan routing and engine composition both read it,
|
|
251
|
+
* so the null and staleness semantics cannot drift between readers. The
|
|
252
|
+
* parameter is typed on the fields the rule reads, not on the full record --
|
|
253
|
+
* the plan's table copy names no `reason`.
|
|
254
|
+
*
|
|
255
|
+
* @param {{available: boolean, exhaustedUntil: string|null, observedAt?: string|null, window?: string|null, [key: string]: unknown}|undefined} availability
|
|
256
|
+
* @param {number} [now] epoch milliseconds; defaults to the current clock
|
|
257
|
+
* @returns {boolean}
|
|
258
|
+
*/
|
|
259
|
+
export function isRuntimeAvailable(availability, now = Date.now()) {
|
|
231
260
|
if (!availability) return false;
|
|
232
|
-
|
|
233
|
-
|
|
261
|
+
const rested = availability.available === true
|
|
262
|
+
? !availability.exhaustedUntil || Date.parse(availability.exhaustedUntil) <= now
|
|
263
|
+
: Boolean(availability.exhaustedUntil && Date.parse(availability.exhaustedUntil) <= now);
|
|
264
|
+
if (!rested) return false;
|
|
265
|
+
const windowSec = availability.window === undefined || availability.window === null
|
|
266
|
+
? undefined
|
|
267
|
+
: AVAILABILITY_WINDOW_SEC[availability.window];
|
|
268
|
+
if (windowSec === undefined || availability.observedAt === undefined || availability.observedAt === null) return true;
|
|
269
|
+
const observedAt = Date.parse(availability.observedAt);
|
|
270
|
+
return !Number.isNaN(observedAt) && observedAt + windowSec * 1000 >= now;
|
|
234
271
|
}
|
|
235
272
|
|
package/src/engine/scheduler.mjs
CHANGED
|
@@ -463,23 +463,23 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
463
463
|
// batched call it used to be, but the chain still runs them one at a time.
|
|
464
464
|
// A settlement may itself dispatch the node's next phase (a judge, a
|
|
465
465
|
// revision) through the same `startJudge`/`startWorker` calls dispatch below
|
|
466
|
-
// uses
|
|
467
|
-
//
|
|
468
|
-
//
|
|
469
|
-
//
|
|
470
|
-
//
|
|
471
|
-
//
|
|
466
|
+
// uses, so it is handed the real `running` to dispatch into: a job it starts
|
|
467
|
+
// is counted against `maxParallel` from the instant the process exists, and
|
|
468
|
+
// `applyRejection` reads that same map to decide whether the run has a slot
|
|
469
|
+
// for the revision at all. It used to dispatch into the throwaway one-entry
|
|
470
|
+
// map instead, copied back only once the settlement returned, which is how
|
|
471
|
+
// run state-location-and-routing-economics-13 came to hold two workers under
|
|
472
|
+
// `maxParallel: 1` on 2026-09-21. What it may *settle* is still only its own
|
|
473
|
+
// node: the one-entry map below is the job, not the run. The node's own
|
|
474
|
+
// steps stay ordered by never starting a second settlement for a node whose
|
|
475
|
+
// first has not yet cleared `pendingSettlements`.
|
|
472
476
|
const settleClosedJobsInBackground = () => {
|
|
473
477
|
for (const [nodeId, job] of [...running]) {
|
|
474
478
|
if (pendingSettlements.has(nodeId) || !job.closed || invocationAlive(job.invocation)) continue;
|
|
475
479
|
running.delete(nodeId);
|
|
476
|
-
const slot = new Map([[nodeId, job]]);
|
|
477
480
|
const settlement = settlementQueue
|
|
478
|
-
.then(() => finalizeClosedJobs(contract, runDir, states,
|
|
479
|
-
.finally(() =>
|
|
480
|
-
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
481
|
-
pendingSettlements.delete(nodeId);
|
|
482
|
-
});
|
|
481
|
+
.then(() => finalizeClosedJobs(contract, runDir, states, new Map([[nodeId, job]]), lock, campaign.path, running))
|
|
482
|
+
.finally(() => pendingSettlements.delete(nodeId));
|
|
483
483
|
// The queue itself must never reject -- a rejected settlement (a lost
|
|
484
484
|
// lock, a programmer error) would otherwise wedge every node queued
|
|
485
485
|
// behind it. The rejection still reaches whoever awaits the real
|
|
@@ -635,20 +635,16 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
635
635
|
// does, on the same per-run candidate ref and worktree
|
|
636
636
|
// `settlementQueue` exists to serialize -- so it is dispatched the
|
|
637
637
|
// same way: chained onto the queue rather than awaited here, using
|
|
638
|
-
// its
|
|
639
|
-
//
|
|
640
|
-
// a time, gated by `pendingSettlements`); only
|
|
641
|
-
// waiting behind it.
|
|
638
|
+
// `running` itself as its dispatch map, so the judge it starts is
|
|
639
|
+
// counted the instant it exists. The node's own order is untouched
|
|
640
|
+
// (still one entry at a time, gated by `pendingSettlements`); only
|
|
641
|
+
// the tick stops waiting behind it.
|
|
642
642
|
if (state.phase === "judge" && state.result) {
|
|
643
643
|
const workerResult = state.result;
|
|
644
|
-
const slot = new Map();
|
|
645
644
|
const settlement = settlementQueue
|
|
646
|
-
.then(() => startJudge(contract, node, state, runDir,
|
|
647
|
-
.then((round) => applyJudgeRound(round, contract, node, state, runDir,
|
|
648
|
-
.finally(() =>
|
|
649
|
-
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
650
|
-
pendingSettlements.delete(node.id);
|
|
651
|
-
});
|
|
645
|
+
.then(() => startJudge(contract, node, state, runDir, running, workerResult, lock, states, campaign.path))
|
|
646
|
+
.then((round) => applyJudgeRound(round, contract, node, state, runDir, running, lock, states, campaign.path, workerResult))
|
|
647
|
+
.finally(() => pendingSettlements.delete(node.id));
|
|
652
648
|
settlementQueue = settlement.catch(() => {});
|
|
653
649
|
settlement.catch((error) => {
|
|
654
650
|
if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
|
package/src/engine/scope.mjs
CHANGED
|
@@ -3,15 +3,19 @@
|
|
|
3
3
|
* touched, and what to do when those differ.
|
|
4
4
|
*
|
|
5
5
|
* Scope is advisory by design -- an unexpected write is recorded as a finding
|
|
6
|
-
* and shown to the judge, not treated as a crime -- with
|
|
6
|
+
* and shown to the judge, not treated as a crime -- with two exceptions:
|
|
7
7
|
* `resolveUnknownEffect` decides whether an invocation whose effect is unproven
|
|
8
|
-
* may be replayed at all, and a dirty scope there is a refusal
|
|
8
|
+
* may be replayed at all, and a dirty scope there is a refusal; and a write
|
|
9
|
+
* that lands on a file the node's own proof names is never deferred, because a
|
|
10
|
+
* verification that passes over an edited prover has proven nothing.
|
|
9
11
|
*/
|
|
12
|
+
import { relative, resolve } from "node:path";
|
|
13
|
+
|
|
10
14
|
import { SETTLED } from "./prompts.mjs";
|
|
11
15
|
import { appendTransitionEvent, recordExecutionOverride, transition, writeNode } from "./state.mjs";
|
|
12
16
|
import { attemptWorkspace } from "../repo/worktree.mjs";
|
|
13
17
|
|
|
14
|
-
import { errorCode, errorMessage, excerpt } from "../util.mjs";
|
|
18
|
+
import { errorCode, errorMessage, excerpt, isContained } from "../util.mjs";
|
|
15
19
|
import { executeControllerVerification } from "./verify.mjs";
|
|
16
20
|
import { providerReceiptsFromInvocationTail, settleInvocation } from "../run/operations.mjs";
|
|
17
21
|
import { readJson } from "../run/store.mjs";
|
|
@@ -98,6 +102,94 @@ export function workerScope(taskPacket) {
|
|
|
98
102
|
roots: taskPacket.writeRoots ?? [],
|
|
99
103
|
};
|
|
100
104
|
}
|
|
105
|
+
/**
|
|
106
|
+
* @typedef {{tokens: string[], cwd: string, literal: boolean, citation: string}} ProofCitation
|
|
107
|
+
*/
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Everything a node's own proofs name: a Definition of Done `path` proof's
|
|
111
|
+
* path, the words of a `command` proof (a command proof carries a display
|
|
112
|
+
* string, not an argv, so a quoted path holding a space is not recovered), the
|
|
113
|
+
* argv of the verification entry a `verification` proof references, and the
|
|
114
|
+
* argv of every verification command the packet declares.
|
|
115
|
+
*
|
|
116
|
+
* @param {ValidatedNode} node
|
|
117
|
+
* @returns {ProofCitation[]}
|
|
118
|
+
*/
|
|
119
|
+
function proofCitations(node) {
|
|
120
|
+
const commands = node.taskPacket.verification ?? [];
|
|
121
|
+
/** @type {ProofCitation[]} */
|
|
122
|
+
const citations = [];
|
|
123
|
+
// Definition of Done items come first so that a path both a checklist item
|
|
124
|
+
// and a verification command name is reported under the checklist item, the
|
|
125
|
+
// name a human reading the failure can act on.
|
|
126
|
+
for (const item of node.definitionOfDone ?? []) {
|
|
127
|
+
const proof = item.proof;
|
|
128
|
+
if (!proof) continue;
|
|
129
|
+
if (proof.kind === "path") {
|
|
130
|
+
citations.push({ tokens: [proof.ref], cwd: ".", literal: true, citation: `${item.id} path proof` });
|
|
131
|
+
} else if (proof.kind === "command") {
|
|
132
|
+
citations.push({ tokens: proof.ref.split(/\s+/u), cwd: ".", literal: false, citation: `${item.id} command proof` });
|
|
133
|
+
} else {
|
|
134
|
+
const command = commands[Number.parseInt(proof.ref, 10)];
|
|
135
|
+
if (command) citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `${item.id} verification[${proof.ref}] proof` });
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
for (const [index, command] of commands.entries()) {
|
|
139
|
+
citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `verification[${index}]` });
|
|
140
|
+
}
|
|
141
|
+
return citations;
|
|
142
|
+
}
|
|
143
|
+
const PATH_SEPARATOR = /[\\/]/u;
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* The unexpected writes that landed on a file the node's own proof names.
|
|
147
|
+
* Matching a command's argv against files is inherently approximate, so this is
|
|
148
|
+
* lexical and deliberately narrow. A word is read as a path only when it is not
|
|
149
|
+
* an option, is shaped like one (a `path` proof's ref, or a word carrying a
|
|
150
|
+
* separator or an extension), and resolves inside the workspace; it then claims
|
|
151
|
+
* an unexpected path it equals, or -- when it carries a separator or is a `path`
|
|
152
|
+
* proof's ref, so a directory really was named -- one it is the directory
|
|
153
|
+
* prefix of.
|
|
154
|
+
*
|
|
155
|
+
* What it deliberately does not catch: a file a proof reaches through a script
|
|
156
|
+
* (`npm test`), a shell string, or a glob the tool expands itself. The bare
|
|
157
|
+
* words of a command are never paths, which is what keeps the ordinary case
|
|
158
|
+
* advisory -- measured against the real node `requirement-ids-reach-the-node`
|
|
159
|
+
* (run `state-location-and-routing-economics-10-requirement-ids-and-closure`),
|
|
160
|
+
* whose legitimate out-of-scope write to `src/plan/freeze.mjs` is claimed by
|
|
161
|
+
* none of its proofs: not by `npm run typecheck`, not by the two test files its
|
|
162
|
+
* `command` proofs name, and not by the loose words of a quoted
|
|
163
|
+
* `--test-name-pattern`.
|
|
164
|
+
*
|
|
165
|
+
* @param {ValidatedNode} node
|
|
166
|
+
* @param {string[]} unexpectedPaths
|
|
167
|
+
* @param {string} workspace
|
|
168
|
+
* @returns {{path: string, citation: string}[]}
|
|
169
|
+
*/
|
|
170
|
+
function proofCitedWrites(node, unexpectedPaths, workspace) {
|
|
171
|
+
/** @type {Map<string, string>} */
|
|
172
|
+
const cited = new Map();
|
|
173
|
+
for (const citation of proofCitations(node)) {
|
|
174
|
+
const base = resolve(workspace, citation.cwd);
|
|
175
|
+
for (const token of citation.tokens) {
|
|
176
|
+
if (!token || token.startsWith("-")) continue;
|
|
177
|
+
const directory = citation.literal || PATH_SEPARATOR.test(token);
|
|
178
|
+
if (!directory && !/\.[A-Za-z0-9]+$/u.test(token)) continue;
|
|
179
|
+
const target = resolve(base, token);
|
|
180
|
+
if (!isContained(workspace, target)) continue;
|
|
181
|
+
const named = relative(workspace, target).replaceAll("\\", "/");
|
|
182
|
+
// The workspace root itself names no file in particular: a proof run from
|
|
183
|
+
// the root must not make every unexpected write a proof-citing one.
|
|
184
|
+
if (!named) continue;
|
|
185
|
+
for (const path of unexpectedPaths) {
|
|
186
|
+
if (cited.has(path)) continue;
|
|
187
|
+
if (path === named || (directory && path.startsWith(`${named}/`))) cited.set(path, citation.citation);
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
return [...cited].map(([path, citation]) => ({ path, citation }));
|
|
192
|
+
}
|
|
101
193
|
/**
|
|
102
194
|
* @param {ValidatedContract} contract
|
|
103
195
|
* @param {string} runDir
|
|
@@ -122,9 +214,19 @@ export function checkWorkerScope(contract, runDir, job, lock, options = {}) {
|
|
|
122
214
|
// A completed attempt whose controller verification passes never fails
|
|
123
215
|
// on scope alone (TECH-SPEC lean, rule 1): the caller defers the verdict
|
|
124
216
|
// until verification has run and records an advisory finding instead.
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
217
|
+
// The single exception is a write onto a file the node's own proof names:
|
|
218
|
+
// verification then passes because the attempt edited the thing doing the
|
|
219
|
+
// proving, and an advisory nobody must read before the gate is too weak a
|
|
220
|
+
// signal for that. Scanned over the bounded path list -- the same first 64
|
|
221
|
+
// paths every other surface reports.
|
|
222
|
+
const cited = proofCitedWrites(job.node, bounded.unexpectedPaths, job.cwd);
|
|
223
|
+
if (options.deferViolation && !cited.length) return true;
|
|
224
|
+
const message = cited.length
|
|
225
|
+
// Kept short on purpose: an error message is capped at 120 characters,
|
|
226
|
+
// and the proof that names the path is the part a reader cannot recover
|
|
227
|
+
// from `state.scope` afterwards.
|
|
228
|
+
? `proof-cited unexpected write (${cited.length}): ${cited.slice(0, 8).map(({ path, citation }) => `${path} (${citation})`).join(", ")}`
|
|
229
|
+
: `unexpected paths changed (${scope.unexpectedPaths.length}): ${bounded.unexpectedPaths.slice(0, 8).join(", ")}`;
|
|
128
230
|
if (!SETTLED.has(state.status)) {
|
|
129
231
|
transition(runDir, state, "failed", { phase: "worker", error: { code: "unexpected_write", message: excerpt(message) } }, lock);
|
|
130
232
|
appendTransitionEvent(runDir, state, "failed", "failed", {
|