faberun 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +193 -31
- package/package.json +1 -1
- package/src/campaign/chain.mjs +19 -0
- package/src/cli.mjs +4 -3
- package/src/contract/final-verification.mjs +41 -10
- package/src/contract/task-packet.mjs +12 -3
- package/src/engine/dispatch.mjs +17 -0
- package/src/engine/lifecycle.mjs +31 -60
- package/src/engine/scheduler.mjs +199 -26
- package/src/engine/settle-judge.mjs +141 -0
- package/src/engine/settle.mjs +1 -1
- package/src/engine/verify.mjs +57 -4
- package/src/harnesses/protocol.mjs +46 -10
- package/src/host/preflight.mjs +19 -0
- package/src/repo/integrate.mjs +6 -1
- package/src/report/render.mjs +58 -58
- package/src/web/index.html +3 -3
- package/src/web/server.mjs +2 -1
package/src/engine/scheduler.mjs
CHANGED
|
@@ -31,15 +31,15 @@ import {
|
|
|
31
31
|
terminalErrorCode,
|
|
32
32
|
} from "./lifecycle.mjs";
|
|
33
33
|
import { delay, errorCode } from "../util.mjs";
|
|
34
|
-
import { alreadyNotified, notifyQueueFor, notifyQueuesByRun, renderCampaignHandoffSafely } from "./notify-queue.mjs";
|
|
35
|
-
import { detectStalls, terminateProcess } from "./process.mjs";
|
|
34
|
+
import { alreadyNotified, emitNodeAdvisories, notifyQueueFor, notifyQueuesByRun, renderCampaignHandoffSafely } from "./notify-queue.mjs";
|
|
35
|
+
import { detectStalls, invocationAlive, terminateProcess } from "./process.mjs";
|
|
36
36
|
import { transition, writeNode } from "./state.mjs";
|
|
37
37
|
import { listNodeSnapshots, readNodeSnapshot } from "../run/node-store.mjs";
|
|
38
38
|
import { render, renderFinalReport, writeFindingsArtifact } from "../report/final.mjs";
|
|
39
39
|
import { operationNextState, providerReceipts, settleInvocation } from "../run/operations.mjs";
|
|
40
40
|
import { appendUsageRecord, invocationCost, invocationUsage, recordInvocationUsage } from "../run/usage.mjs";
|
|
41
41
|
import { captureNodeScopeBoundaries, checkWorkerScope, emptyScope } from "./scope.mjs";
|
|
42
|
-
import {
|
|
42
|
+
import { validateContractForLaunch } from "../campaign/chain.mjs";
|
|
43
43
|
import { validateNodeSnapshot } from "../contract/snapshot.mjs";
|
|
44
44
|
import { finalVerificationCommands, sharedVerificationCommands } from "../contract/final-verification.mjs";
|
|
45
45
|
import { startJudge, startWorker } from "./dispatch.mjs";
|
|
@@ -131,16 +131,22 @@ export function nodeBudgetBasisMs(contract, node) {
|
|
|
131
131
|
* never touches phase 2's parked `blocked`/`failed`/`exhausted`/`stalled`
|
|
132
132
|
* states, and never overwrites the `runtime_tier_exhausted` waiting shape.
|
|
133
133
|
*
|
|
134
|
+
* A node whose closed job is being settled in the background (its id is a key
|
|
135
|
+
* of `pendingSettlements`) is not this dead end either: its invocation has
|
|
136
|
+
* already exited and left `running`, but the settlement promise still owns
|
|
137
|
+
* deciding what happens to it, so it is left alone until that promise resolves.
|
|
138
|
+
*
|
|
134
139
|
* @param {string} runDir
|
|
135
140
|
* @param {Map<string, NodeSnapshot>} states
|
|
136
141
|
* @param {Map<string, Job>} running
|
|
137
142
|
* @param {LockHandle|null} lock
|
|
143
|
+
* @param {Map<string, Promise<void>>} [pendingSettlements]
|
|
138
144
|
* @returns {string[]} the node ids this pass parked
|
|
139
145
|
*/
|
|
140
|
-
export function enforceRunningInvariant(runDir, states, running, lock) {
|
|
146
|
+
export function enforceRunningInvariant(runDir, states, running, lock, pendingSettlements = new Map()) {
|
|
141
147
|
const parked = [];
|
|
142
148
|
for (const [nodeId, state] of states) {
|
|
143
|
-
if (state.status !== "running" || running.has(nodeId)) continue;
|
|
149
|
+
if (state.status !== "running" || running.has(nodeId) || pendingSettlements.has(nodeId)) continue;
|
|
144
150
|
transition(runDir, state, "blocked", {
|
|
145
151
|
phase: state.phase,
|
|
146
152
|
error: {
|
|
@@ -175,14 +181,17 @@ function renderFingerprint(states) {
|
|
|
175
181
|
|
|
176
182
|
/**
|
|
177
183
|
* @param {string} contractPath
|
|
178
|
-
* @param {{detachedBootstrap?: boolean}} [options]
|
|
179
|
-
* only by the CLI entry when this process is its
|
|
180
|
-
* makes the controller wait for the launcher's
|
|
184
|
+
* @param {{detachedBootstrap?: boolean, baseRef?: string}} [options]
|
|
185
|
+
* `detachedBootstrap` is set only by the CLI entry when this process is its
|
|
186
|
+
* own detached child, and makes the controller wait for the launcher's
|
|
187
|
+
* acknowledgement; `baseRef` is the CLI's own `--base-ref`, re-validated
|
|
188
|
+
* here so a launch and the run it starts agree about what the contract was
|
|
189
|
+
* checked against
|
|
181
190
|
* @returns {Promise<RunOutcome>}
|
|
182
191
|
*/
|
|
183
192
|
export async function runContract(contractPath, options = {}) {
|
|
184
193
|
const absoluteContractPath = resolve(contractPath);
|
|
185
|
-
const contract =
|
|
194
|
+
const contract = validateContractForLaunch(JSON.parse(readFileSync(absoluteContractPath, "utf8")), absoluteContractPath, { baseRef: options.baseRef });
|
|
186
195
|
const runDir = join(contract.cwd, ".runs", contract.id);
|
|
187
196
|
if (existsSync(runDir)) throw new Error(`run already exists: ${runDir}`);
|
|
188
197
|
mkdirSync(join(contract.cwd, ".runs"), { recursive: true });
|
|
@@ -319,12 +328,13 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
319
328
|
statusFingerprint = fingerprint;
|
|
320
329
|
render(runDir, runsDir, contract, states, renderLock);
|
|
321
330
|
};
|
|
322
|
-
//
|
|
323
|
-
//
|
|
324
|
-
//
|
|
325
|
-
//
|
|
326
|
-
//
|
|
327
|
-
// writes nothing
|
|
331
|
+
// A node's own settlement (its controller verification, its judge round) can
|
|
332
|
+
// be minutes long, and it now runs off the tick's critical path in
|
|
333
|
+
// `pendingSettlements` so dispatch never waits behind it -- but that whole
|
|
334
|
+
// time status.json would otherwise report whatever the last tick left it at.
|
|
335
|
+
// A timer renders between ticks too; it is cheap even when idle because
|
|
336
|
+
// `renderFingerprint` still change-detects, so a quiet run writes nothing
|
|
337
|
+
// extra.
|
|
328
338
|
const statusTimer = setInterval(() => renderStatusIfChanged(), contract.pollIntervalMs);
|
|
329
339
|
statusTimer.unref();
|
|
330
340
|
let handoffFingerprint = statesFingerprint(states);
|
|
@@ -364,14 +374,31 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
364
374
|
|
|
365
375
|
/** @type {Map<string, Job>} */
|
|
366
376
|
const running = new Map();
|
|
377
|
+
// One promise per node currently settling a closed job -- its controller
|
|
378
|
+
// verification, its candidate verification, its judge round -- kept off the
|
|
379
|
+
// tick's critical path so an eligible sibling still dispatches into a free
|
|
380
|
+
// slot while this node's own invocation has already exited. A node's own
|
|
381
|
+
// steps stay ordered because only one settlement per node is ever in flight
|
|
382
|
+
// (see the dispatch loop below); a *different* node's settlement is queued
|
|
383
|
+
// behind whichever one is already running (`settlementQueue`), not run
|
|
384
|
+
// alongside it -- `finalizeClosedJobs` shares state a concurrent second call
|
|
385
|
+
// would corrupt: the phase-continuation selection that picks at most one
|
|
386
|
+
// live node to carry a session forward, and `repo/integrate.mjs`'s one
|
|
387
|
+
// candidate ref and worktree per run. Only *dispatching a sibling* skips
|
|
388
|
+
// ahead of a node's settlement; two nodes' settlements never interleave.
|
|
389
|
+
/** @type {Map<string, Promise<void>>} */
|
|
390
|
+
const pendingSettlements = new Map();
|
|
391
|
+
/** @type {Promise<void>} */
|
|
392
|
+
let settlementQueue = Promise.resolve();
|
|
367
393
|
let canceled = false;
|
|
368
394
|
const cancel = () => { canceled = true; };
|
|
369
395
|
process.once("SIGINT", cancel);
|
|
370
396
|
process.once("SIGTERM", cancel);
|
|
371
397
|
process.once("SIGHUP", cancel);
|
|
372
398
|
// The heartbeat's `at` is owned by an unref'd timer inside this writer, never
|
|
373
|
-
// by the loop body below:
|
|
374
|
-
// critical path and
|
|
399
|
+
// by the loop body below: a node's settlement can still run for minutes off
|
|
400
|
+
// the tick's critical path (`pendingSettlements`), and the heartbeat must
|
|
401
|
+
// keep answering "the process is alive" while it does.
|
|
375
402
|
const heartbeat = createHeartbeat({ runDir, intervalMs: HEARTBEAT_INTERVAL_MS });
|
|
376
403
|
const heartbeatNodes = new Map(contract.nodes.map((node) => [node.id, node]));
|
|
377
404
|
let heartbeatFingerprint = statesFingerprint(states);
|
|
@@ -380,11 +407,89 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
380
407
|
const activeHeartbeatNodes = () => [...states.values()]
|
|
381
408
|
.filter((state) => state.status === "running" && heartbeatNodes.has(state.id))
|
|
382
409
|
.map((state) => ({ nodeId: state.id, budgetBasis: nodeBudgetBasisMs(contract, /** @type {ValidatedNode} */ (heartbeatNodes.get(state.id))) }));
|
|
410
|
+
// A programmer error surfacing inside a background settlement must still
|
|
411
|
+
// crash the whole run, exactly as an unguarded `await finalizeClosedJobs`
|
|
412
|
+
// used to: it is recorded here and thrown from the top of the loop on the
|
|
413
|
+
// very next tick, rather than immediately, so it cannot itself become the
|
|
414
|
+
// block a sibling's dispatch is waiting behind. A lost lock is not this --
|
|
415
|
+
// the loop's own `lock.assert()` calls surface that same condition on their
|
|
416
|
+
// own schedule, so it is left for them.
|
|
417
|
+
/** @type {unknown} */
|
|
418
|
+
let backgroundSettlementFailure = null;
|
|
419
|
+
// Mark every job that closed this tick as settling, and free its slot,
|
|
420
|
+
// without waiting for any of them: that alone is what lets an eligible
|
|
421
|
+
// sibling dispatch into the freed slot while this node's minutes-long
|
|
422
|
+
// controller verification or judge round is still running. The actual
|
|
423
|
+
// settlement work is chained onto `settlementQueue`, one node at a time in
|
|
424
|
+
// the order its job closed, so it still runs exactly as serialized against
|
|
425
|
+
// every *other* node's settlement as it did when this loop awaited
|
|
426
|
+
// `finalizeClosedJobs` directly -- `finalizeClosedJobs` runs against a
|
|
427
|
+
// one-entry map per node, so this is one call per node rather than the one
|
|
428
|
+
// batched call it used to be, but the chain still runs them one at a time.
|
|
429
|
+
// A settlement may itself dispatch the node's next phase (a judge, a
|
|
430
|
+
// revision) through the same `startJudge`/`startWorker` calls dispatch below
|
|
431
|
+
// uses; those land in `slot`, not the real `running`, so they are copied
|
|
432
|
+
// back into `running` in a `finally` -- unconditionally, win or lose, so a
|
|
433
|
+
// job a settlement started before failing (a lost lock, a programmer error)
|
|
434
|
+
// is still visible to the cleanup sweeps below rather than leaked. The
|
|
435
|
+
// node's own steps stay ordered by never starting a second settlement for a
|
|
436
|
+
// node whose first has not yet cleared `pendingSettlements`.
|
|
437
|
+
const settleClosedJobsInBackground = () => {
|
|
438
|
+
for (const [nodeId, job] of [...running]) {
|
|
439
|
+
if (pendingSettlements.has(nodeId) || !job.closed || invocationAlive(job.invocation)) continue;
|
|
440
|
+
running.delete(nodeId);
|
|
441
|
+
const slot = new Map([[nodeId, job]]);
|
|
442
|
+
const settlement = settlementQueue
|
|
443
|
+
.then(() => finalizeClosedJobs(contract, runDir, states, slot, lock, campaign.path))
|
|
444
|
+
.finally(() => {
|
|
445
|
+
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
446
|
+
pendingSettlements.delete(nodeId);
|
|
447
|
+
});
|
|
448
|
+
// The queue itself must never reject -- a rejected settlement (a lost
|
|
449
|
+
// lock, a programmer error) would otherwise wedge every node queued
|
|
450
|
+
// behind it. The rejection still reaches whoever awaits the real
|
|
451
|
+
// `settlement` promise (`pendingSettlements`, below).
|
|
452
|
+
settlementQueue = settlement.catch(() => {});
|
|
453
|
+
// Handled here so an in-flight settlement never becomes an unhandled
|
|
454
|
+
// rejection when nobody happens to await `pendingSettlements` before the
|
|
455
|
+
// process exits; the original promise, still held below, carries the
|
|
456
|
+
// rejection to whichever checkpoint (cancel, shutdown) awaits it.
|
|
457
|
+
settlement.catch((error) => {
|
|
458
|
+
if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
|
|
459
|
+
});
|
|
460
|
+
pendingSettlements.set(nodeId, settlement);
|
|
461
|
+
}
|
|
462
|
+
};
|
|
463
|
+
// Captured once, before the loop, rather than re-derived every tick: it is
|
|
464
|
+
// "parked when this controller invocation started" (autoRetryParkedNodes's
|
|
465
|
+
// own contract), and a node a background settlement parks between two ticks
|
|
466
|
+
// -- rather than synchronously within one, as it always did before
|
|
467
|
+
// settlement moved off the tick's critical path -- must still read as newly
|
|
468
|
+
// parked whichever later tick first observes it, not as already-parked
|
|
469
|
+
// because a per-tick snapshot happened to be taken after it landed.
|
|
470
|
+
const parkedBefore = new Set([...states.values()].filter((state) => PARKED.has(state.status)).map((state) => state.id));
|
|
471
|
+
const anyUnsettled = () => [...states.values()].some((state) => !SETTLED.has(state.status));
|
|
383
472
|
try {
|
|
384
|
-
while
|
|
473
|
+
// A `while` that re-checked this at the very top of every tick would exit
|
|
474
|
+
// the instant a background settlement flips the run's last unsettled node
|
|
475
|
+
// straight to a terminal status between two ticks -- before the tick body
|
|
476
|
+
// that would have run `autoRetryParkedNodes` against that new status ever
|
|
477
|
+
// gets to. The entry guard skips the loop entirely when there is nothing
|
|
478
|
+
// to do at all (a resume of an already-settled run still does zero
|
|
479
|
+
// iterations); once inside, the exit check moves to the bottom, after the
|
|
480
|
+
// body, so that body always sees a freshly-parked node at least once
|
|
481
|
+
// before the loop is allowed to end.
|
|
482
|
+
if (anyUnsettled()) for (;;) {
|
|
385
483
|
lock.assert();
|
|
484
|
+
if (backgroundSettlementFailure !== null) throw backgroundSettlementFailure;
|
|
386
485
|
if (existsSync(join(runDir, "cancel.request.json"))) canceled = true;
|
|
387
486
|
if (canceled) {
|
|
487
|
+
// A settlement already in flight owns the one decision a cancellation
|
|
488
|
+
// must not race: what this node's own invocation resolved to. Let it
|
|
489
|
+
// land on its real terminal status (and, when it re-dispatched a judge
|
|
490
|
+
// or a revision, on the new job that landed in `running`) before this
|
|
491
|
+
// branch decides which nodes are merely canceled.
|
|
492
|
+
await Promise.all([...pendingSettlements.values()]);
|
|
388
493
|
const jobs = [...running.values()];
|
|
389
494
|
await Promise.all(jobs.map((job) => terminateProcess(job)));
|
|
390
495
|
const envelopes = new Map();
|
|
@@ -408,8 +513,14 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
408
513
|
break;
|
|
409
514
|
}
|
|
410
515
|
|
|
411
|
-
|
|
412
|
-
|
|
516
|
+
// Advisory spend lines are checked every tick, closed job or not: a
|
|
517
|
+
// crossing must be visible while the spend is happening on a node still
|
|
518
|
+
// running, not only once it closes. `finalizeClosedJobs` used to open
|
|
519
|
+
// with this same check; calling it once here, rather than once per
|
|
520
|
+
// node settled this tick, is what keeps it at one pass per tick now
|
|
521
|
+
// that settlement runs per node in `settleClosedJobsInBackground`.
|
|
522
|
+
await emitNodeAdvisories(contract, runDir, states);
|
|
523
|
+
settleClosedJobsInBackground();
|
|
413
524
|
await detectStalls(contract, running, async (job, status, error) => {
|
|
414
525
|
const envelope = recordInvocationUsage(job);
|
|
415
526
|
job.state.usage = invocationUsage(job.state);
|
|
@@ -450,8 +561,11 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
450
561
|
writeNode(runDir, job.state, lock);
|
|
451
562
|
});
|
|
452
563
|
// A node left `running` with no job is a dead end; park it before the
|
|
453
|
-
// dispatch pass so it cannot hide behind a healthy sibling.
|
|
454
|
-
|
|
564
|
+
// dispatch pass so it cannot hide behind a healthy sibling. A node
|
|
565
|
+
// whose closed job is settling in the background is not that dead end
|
|
566
|
+
// -- `pendingSettlements` is what tells this apart from one truly
|
|
567
|
+
// abandoned.
|
|
568
|
+
enforceRunningInvariant(runDir, states, running, lock, pendingSettlements);
|
|
455
569
|
// Before dependants are blocked, a node that parked on this tick gets
|
|
456
570
|
// its one automatic retry: it becomes pending, so `blockDependents` sees
|
|
457
571
|
// nothing to block and the dependants stay `pending`/`phase: "waiting"`
|
|
@@ -463,14 +577,37 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
463
577
|
if (slots > 0) {
|
|
464
578
|
const ready = contract.nodes.filter((node) => {
|
|
465
579
|
const state = states.get(node.id);
|
|
466
|
-
return state?.status === "pending" &&
|
|
580
|
+
return state?.status === "pending" && !pendingSettlements.has(node.id)
|
|
581
|
+
&& node.dependsOn.every((id) => states.get(id)?.status === "done");
|
|
467
582
|
});
|
|
468
583
|
for (const node of ready.slice(0, slots)) {
|
|
469
584
|
const state = states.get(node.id);
|
|
470
585
|
if (!state || routingBackoffActive(state, state.phase)) continue;
|
|
586
|
+
// A node recovered pending a re-ask judge (its own worker attempt
|
|
587
|
+
// already accepted, `state.result` durable) reaches `settleDone` /
|
|
588
|
+
// `integrateAttempt` exactly like a closed job's own settlement
|
|
589
|
+
// does, on the same per-run candidate ref and worktree
|
|
590
|
+
// `settlementQueue` exists to serialize -- so it is dispatched the
|
|
591
|
+
// same way: chained onto the queue rather than awaited here, using
|
|
592
|
+
// its own one-entry `slot` merged back into `running` in a
|
|
593
|
+
// `finally`. The node's own order is untouched (still one entry at
|
|
594
|
+
// a time, gated by `pendingSettlements`); only the tick stops
|
|
595
|
+
// waiting behind it.
|
|
471
596
|
if (state.phase === "judge" && state.result) {
|
|
472
|
-
|
|
473
|
-
|
|
597
|
+
const workerResult = state.result;
|
|
598
|
+
const slot = new Map();
|
|
599
|
+
const settlement = settlementQueue
|
|
600
|
+
.then(() => startJudge(contract, node, state, runDir, slot, workerResult, lock, states, campaign.path))
|
|
601
|
+
.then((round) => applyJudgeRound(round, contract, node, state, runDir, slot, lock, states, campaign.path, workerResult))
|
|
602
|
+
.finally(() => {
|
|
603
|
+
for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
|
|
604
|
+
pendingSettlements.delete(node.id);
|
|
605
|
+
});
|
|
606
|
+
settlementQueue = settlement.catch(() => {});
|
|
607
|
+
settlement.catch((error) => {
|
|
608
|
+
if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
|
|
609
|
+
});
|
|
610
|
+
pendingSettlements.set(node.id, settlement);
|
|
474
611
|
continue;
|
|
475
612
|
}
|
|
476
613
|
state.attempt += 1;
|
|
@@ -499,10 +636,46 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
|
|
|
499
636
|
}
|
|
500
637
|
}
|
|
501
638
|
heartbeat.setActive(activeHeartbeatNodes());
|
|
502
|
-
|
|
639
|
+
// A background settlement can flip a node straight to a parked
|
|
640
|
+
// terminal status during this same tick's own later awaits
|
|
641
|
+
// (`detectStalls`, `notifyStateChanges`), after the auto-retry pass
|
|
642
|
+
// above already ran and saw it as not-yet-parked. Running it once more
|
|
643
|
+
// here, immediately before the exit check, is what stops the loop from
|
|
644
|
+
// mistaking that freshly-parked node for settled and exiting before it
|
|
645
|
+
// ever got its automatic retry; a node it reopens dispatches on the
|
|
646
|
+
// next tick rather than this one, which `anyUnsettled()` below still
|
|
647
|
+
// correctly keeps the loop alive for.
|
|
648
|
+
autoRetryParkedNodes(contract, runDir, states, lock, parkedBefore);
|
|
649
|
+
if (!anyUnsettled()) break;
|
|
650
|
+
await delay(contract.pollIntervalMs);
|
|
503
651
|
}
|
|
652
|
+
// The loop above exits (by the bottom check above or the cancel branch's
|
|
653
|
+
// `break`) the instant every node's state looks settled, but a
|
|
654
|
+
// settlement's own trailing work -- sealing a candidate's acceptance,
|
|
655
|
+
// removing its worktree, enqueueing its terminal notification -- can
|
|
656
|
+
// still be running after the transition that made the state look
|
|
657
|
+
// terminal. Nothing past this point (the final render, the findings
|
|
658
|
+
// artifact, the run-terminal notification, releasing the lock) may run
|
|
659
|
+
// ahead of that trailing work.
|
|
660
|
+
await Promise.all([...pendingSettlements.values()]);
|
|
661
|
+
// A settlement's own `finally` deletes it from `pendingSettlements` the
|
|
662
|
+
// instant it settles, win or lose -- so a rejection recorded here can
|
|
663
|
+
// already be gone from the map above by the time this line runs, with
|
|
664
|
+
// nothing left to await it. The loop's own top-of-tick check
|
|
665
|
+
// (`if (backgroundSettlementFailure !== null) throw ...`) cannot save
|
|
666
|
+
// that case either: the loop has already exited. Checked again here,
|
|
667
|
+
// once, so a background settlement failure can never read as a clean
|
|
668
|
+
// finish just because it settled on the same tick the run's last node
|
|
669
|
+
// did.
|
|
670
|
+
if (backgroundSettlementFailure !== null) throw backgroundSettlementFailure;
|
|
504
671
|
} catch (error) {
|
|
505
672
|
if (!(error instanceof LockLostError)) throw error;
|
|
673
|
+
// A settlement still merges whatever it started into `running` in its own
|
|
674
|
+
// `finally`, win or lose, regardless of whether anything awaits it; wait
|
|
675
|
+
// for that to land (never rejecting itself, so a second lock-loss here
|
|
676
|
+
// cannot mask the one already being handled) before the termination sweep
|
|
677
|
+
// reads `running`.
|
|
678
|
+
await Promise.allSettled([...pendingSettlements.values()]);
|
|
506
679
|
await Promise.all([...running.values()].map((job) => terminateProcess(job)));
|
|
507
680
|
notifyQueuesByRun.delete(runDir);
|
|
508
681
|
return { runDir, states, ok: false, error };
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The judge branch of `finalizeClosedJobs`, and the workspace comparison its
|
|
3
|
+
* fail-closed write check depends on. Split out of `lifecycle.mjs` for the
|
|
4
|
+
* same reason `engine/settle.mjs` was: `lifecycle.mjs` and
|
|
5
|
+
* `test/engine/judge.test.mjs` both sit on the 800-line ceiling
|
|
6
|
+
* `test/repo/source-shape.test.mjs` enforces.
|
|
7
|
+
*
|
|
8
|
+
* `settleJudgeRound` runs only after `lifecycle.mjs` has already called
|
|
9
|
+
* `judgeWorkspaceWriteViolation` and found nothing: the check has to run
|
|
10
|
+
* before any branch here (a failed provider, a bounded re-dispatch, or a
|
|
11
|
+
* verdict) can adopt or launder a judge's own write, so it cannot live inside
|
|
12
|
+
* this function without reintroducing the escape it closes.
|
|
13
|
+
*
|
|
14
|
+
* `clearTierExhaustion` and `handleProviderExhaustion` are `lifecycle.mjs`'s
|
|
15
|
+
* own -- importing them back from there would recreate the exact
|
|
16
|
+
* `lifecycle.mjs` <-> `review.mjs` cycle `test/repo/source-shape.test.mjs`
|
|
17
|
+
* once caught and `AGENTS.md` records as fixed, so the caller passes them in
|
|
18
|
+
* instead.
|
|
19
|
+
*/
|
|
20
|
+
import { judgeReaskOutstanding } from "./judge-gate.mjs";
|
|
21
|
+
import { judgeVerdictEvidence } from "../contract/review-modes.mjs";
|
|
22
|
+
import {
|
|
23
|
+
JUDGE_MAX_FAILURES,
|
|
24
|
+
applyJudgeProtocolFailure,
|
|
25
|
+
applyJudgeResult,
|
|
26
|
+
applyJudgeRound,
|
|
27
|
+
settleUnavailableJudge,
|
|
28
|
+
} from "./review.mjs";
|
|
29
|
+
import { networkTransition } from "./backoff.mjs";
|
|
30
|
+
import { startJudge } from "./dispatch.mjs";
|
|
31
|
+
import { writeNode } from "./state.mjs";
|
|
32
|
+
import { compareWorkspaceSnapshot } from "../repo/workspace.mjs";
|
|
33
|
+
import { errorCode, errorMessage, excerpt } from "../util.mjs";
|
|
34
|
+
|
|
35
|
+
/** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
|
|
36
|
+
/** @typedef {import("../contract/index.mjs").NodeSnapshot} NodeSnapshot */
|
|
37
|
+
/** @typedef {import("./process.mjs").Job} Job */
|
|
38
|
+
/** @typedef {import("../run/lock.mjs").LockRecord} LockRecord */
|
|
39
|
+
/** @typedef {ReturnType<typeof import("../run/lock.mjs").acquire>} LockHandle */
|
|
40
|
+
/** @typedef {import("../harnesses/index.mjs").ProviderEnvelope} ProviderEnvelope */
|
|
41
|
+
/** @typedef {ProviderEnvelope & {costProvenance?: "priced"}} PricedEnvelope */
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Whether a closed judge job wrote into its own workspace, or a workspace
|
|
45
|
+
* comparison error that means the same thing: either is a violation, never
|
|
46
|
+
* silence. `compareWorkspaceSnapshot` throws `snapshot_ignore_changed` when
|
|
47
|
+
* the judge edits an ignore source, `snapshot_symlink_escape` when it creates
|
|
48
|
+
* a symlink out of the tree, and `snapshot_too_large` when it adds enough
|
|
49
|
+
* entries -- all ordinary judge writes that must fail closed exactly like the
|
|
50
|
+
* worker path (`engine/scope.mjs`'s `checkWorkerScope`), never vanish into a
|
|
51
|
+
* "no writes" result. `job.scopeBaseline` is null only when the capture at
|
|
52
|
+
* dispatch time itself failed -- `startJudge` degrades to unchecked rather
|
|
53
|
+
* than refuse to run the judge at all -- and that is the one case with
|
|
54
|
+
* nothing to compare, not a violation.
|
|
55
|
+
*
|
|
56
|
+
* @param {Job} job
|
|
57
|
+
* @returns {{message: string}|null}
|
|
58
|
+
*/
|
|
59
|
+
export function judgeWorkspaceWriteViolation(job) {
|
|
60
|
+
const baseline = job.scopeBaseline;
|
|
61
|
+
if (!baseline) return null;
|
|
62
|
+
try {
|
|
63
|
+
const { unexpectedPaths } = compareWorkspaceSnapshot(/** @type {import("../repo/workspace.mjs").WorkspaceSnapshot} */ (baseline), job.cwd);
|
|
64
|
+
if (!unexpectedPaths.length) return null;
|
|
65
|
+
return { message: excerpt(`judge wrote into its own workspace (${unexpectedPaths.length}): ${unexpectedPaths.slice(0, 8).join(", ")}`) ?? "judge wrote into its own workspace" };
|
|
66
|
+
} catch (error) {
|
|
67
|
+
return { message: excerpt(`judge workspace comparison failed (${errorCode(error) ?? "scope_snapshot_invalid"}): ${errorMessage(error)}`) ?? "judge workspace comparison failed" };
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Settle a closed judge invocation whose write check already passed: a
|
|
73
|
+
* provider that failed outright gets one bounded re-dispatch, then review-mode
|
|
74
|
+
* settlement; anything else that is not exactly one usable verdict (none at
|
|
75
|
+
* all, several of them, an unparseable one, a stream cut off before its
|
|
76
|
+
* terminal envelope, or a phase killed on its wall clock) takes the same
|
|
77
|
+
* bounded re-ask; a clean verdict is applied.
|
|
78
|
+
*
|
|
79
|
+
* @param {ValidatedContract} contract
|
|
80
|
+
* @param {Job} job
|
|
81
|
+
* @param {NodeSnapshot} state
|
|
82
|
+
* @param {string} runDir
|
|
83
|
+
* @param {Map<string, Job>} running
|
|
84
|
+
* @param {LockHandle} lock
|
|
85
|
+
* @param {Map<string, NodeSnapshot>} states
|
|
86
|
+
* @param {string} campaignPath
|
|
87
|
+
* @param {PricedEnvelope} envelope
|
|
88
|
+
* @param {{clearTierExhaustion: (state: NodeSnapshot) => void, handleProviderExhaustion: typeof import("./lifecycle.mjs").handleProviderExhaustion}} hooks
|
|
89
|
+
* `lifecycle.mjs`'s own tier-exhaustion clear and provider-exhaustion router, passed in rather than imported back.
|
|
90
|
+
* @returns {Promise<void>}
|
|
91
|
+
*/
|
|
92
|
+
export async function settleJudgeRound(contract, job, state, runDir, running, lock, states, campaignPath, envelope, { clearTierExhaustion, handleProviderExhaustion }) {
|
|
93
|
+
const node = job.node;
|
|
94
|
+
// A judge provider that failed outright (its turn died, its tool host was
|
|
95
|
+
// gone) is a provider failure, never a verdict: the gate cannot adopt a
|
|
96
|
+
// result the judge could not ground in inspection. Re-dispatch the judge
|
|
97
|
+
// once on the same routing, then settle by review mode so a judge failure
|
|
98
|
+
// is surfaced, never silently settled. A stream that never reached its
|
|
99
|
+
// terminal envelope is a protocol defect instead and takes the bounded
|
|
100
|
+
// re-ask below.
|
|
101
|
+
if (envelope.status === "failed" && envelope.error?.code !== "incomplete_stream") {
|
|
102
|
+
// A judge that lost its socket is not an unavailable judge. It buys the
|
|
103
|
+
// same bounded network waits a worker does, on the runtime it already
|
|
104
|
+
// warmed, and spends none of the one re-dispatch counted below.
|
|
105
|
+
const network = networkTransition(contract, node, state, "judge", envelope, job.exitCode);
|
|
106
|
+
if (network && handleProviderExhaustion(contract, runDir, node, state, "judge", envelope, job.runtime.id, lock, states, campaignPath, network)) return;
|
|
107
|
+
clearTierExhaustion(state);
|
|
108
|
+
// The provider died on the bounded re-ask itself, so the one permitted
|
|
109
|
+
// re-ask is spent: settle by review mode here rather than dispatch a
|
|
110
|
+
// third judge invocation behind a fresh failure count.
|
|
111
|
+
if (judgeReaskOutstanding(state)) {
|
|
112
|
+
await applyJudgeProtocolFailure(contract, node, state, runDir, running, lock, states, campaignPath, envelope.error?.message ?? "judge provider failed");
|
|
113
|
+
return;
|
|
114
|
+
}
|
|
115
|
+
state.judgeFailures = (state.judgeFailures ?? 0) + 1;
|
|
116
|
+
if (state.judgeFailures < JUDGE_MAX_FAILURES) {
|
|
117
|
+
writeNode(runDir, state, lock);
|
|
118
|
+
await applyJudgeRound(await startJudge(contract, node, state, runDir, running, state.result, lock, states, campaignPath),
|
|
119
|
+
contract, node, state, runDir, running, lock, states, campaignPath, state.result);
|
|
120
|
+
return;
|
|
121
|
+
}
|
|
122
|
+
await settleUnavailableJudge(contract, node, state, runDir, lock, states, campaignPath, envelope.error?.message ?? "judge provider failed");
|
|
123
|
+
return;
|
|
124
|
+
}
|
|
125
|
+
// Whatever else this invocation produced, it is not exactly one usable
|
|
126
|
+
// verdict: no verdict at all, several of them in separate agent messages,
|
|
127
|
+
// an unparseable one, a stream cut off before its terminal envelope, or a
|
|
128
|
+
// phase killed on its wall clock. One bounded re-ask, then the review mode
|
|
129
|
+
// decides — advisory completes, blocking enters attention with the work
|
|
130
|
+
// preserved so a retry in place can re-judge it.
|
|
131
|
+
const evidence = judgeVerdictEvidence(envelope);
|
|
132
|
+
if (!evidence.ok) {
|
|
133
|
+
const network = networkTransition(contract, node, state, "judge", envelope, job.exitCode);
|
|
134
|
+
if (network && handleProviderExhaustion(contract, runDir, node, state, "judge", envelope, job.runtime.id, lock, states, campaignPath, network)) return;
|
|
135
|
+
clearTierExhaustion(state);
|
|
136
|
+
await applyJudgeProtocolFailure(contract, node, state, runDir, running, lock, states, campaignPath, evidence.reason);
|
|
137
|
+
return;
|
|
138
|
+
}
|
|
139
|
+
clearTierExhaustion(state);
|
|
140
|
+
await applyJudgeResult(contract, node, state, evidence.result, runDir, lock, running, states, campaignPath);
|
|
141
|
+
}
|
package/src/engine/settle.mjs
CHANGED
|
@@ -165,7 +165,7 @@ export async function settleDone(contract, node, state, runDir, lock, states, ca
|
|
|
165
165
|
attemptSha: sealed.sha,
|
|
166
166
|
branch: state.worktree.branch,
|
|
167
167
|
verificationEvidence: state.verification,
|
|
168
|
-
verifyCandidate: (candidateWorkspace) => verifyCandidateWorkspace(contract, node, state, runDir, candidateWorkspace),
|
|
168
|
+
verifyCandidate: (candidateWorkspace) => verifyCandidateWorkspace(contract, node, state, runDir, candidateWorkspace, lock),
|
|
169
169
|
onAccepted: async (transaction) => {
|
|
170
170
|
const acceptedPath = state.worktree?.path ?? attemptWorktreePath(runDir, contract.id, node.id, transaction.attempt);
|
|
171
171
|
if (state.attempt === transaction.attempt && state.status !== "done") {
|
package/src/engine/verify.mjs
CHANGED
|
@@ -11,8 +11,10 @@
|
|
|
11
11
|
import { attemptWorkspace } from "../repo/worktree.mjs";
|
|
12
12
|
import { boundedUtf8, errorMessage } from "../util.mjs";
|
|
13
13
|
import { compactVerification } from "../contract/verification.mjs";
|
|
14
|
-
import { finalVerificationCommands, sharedVerificationCommands } from "../contract/final-verification.mjs";
|
|
14
|
+
import { finalVerificationCommands, phaseTerminalNode, sharedVerificationCommands } from "../contract/final-verification.mjs";
|
|
15
15
|
import { join } from "node:path";
|
|
16
|
+
import { listNodeSnapshots, readNodeSnapshot } from "../run/node-store.mjs";
|
|
17
|
+
import { SETTLED } from "./prompts.mjs";
|
|
16
18
|
import { terminateInvocation } from "./process.mjs";
|
|
17
19
|
import { writeNode } from "./state.mjs";
|
|
18
20
|
import { runVerification } from "./run-command.mjs";
|
|
@@ -28,6 +30,34 @@ import { runVerification } from "./run-command.mjs";
|
|
|
28
30
|
/** @typedef {import("../contract/index.mjs").VerificationState} VerificationState */
|
|
29
31
|
/** @typedef {{index: number, total: number, argv: string}} VerificationProgress */
|
|
30
32
|
|
|
33
|
+
/**
|
|
34
|
+
* The sibling phase-terminal node ids already settled, read straight off
|
|
35
|
+
* disk: every node snapshot is persisted from the run's first tick (see
|
|
36
|
+
* `runContract`), so a node still `pending` reads back honestly, not as
|
|
37
|
+
* missing. `finalVerificationCommands` uses this to decide, at the moment a
|
|
38
|
+
* phase-terminal node's own verification runs, whether it is the one that
|
|
39
|
+
* closes the phase -- recomputed fresh on every call, so a retried node sees
|
|
40
|
+
* its siblings' current status each time, not a decision frozen from an
|
|
41
|
+
* earlier attempt.
|
|
42
|
+
*
|
|
43
|
+
* @param {string} runDir
|
|
44
|
+
* @param {ValidatedContract} contract
|
|
45
|
+
* @param {ValidatedNode} node
|
|
46
|
+
* @returns {Set<string>}
|
|
47
|
+
*/
|
|
48
|
+
function settledSiblingIds(runDir, contract, node) {
|
|
49
|
+
const ids = new Set();
|
|
50
|
+
for (const name of listNodeSnapshots(runDir)) {
|
|
51
|
+
const id = name.slice(0, -".json".length);
|
|
52
|
+
if (id === node.id) continue;
|
|
53
|
+
const sibling = contract.nodes.find((candidate) => candidate.id === id);
|
|
54
|
+
if (!sibling || !phaseTerminalNode(contract, sibling)) continue;
|
|
55
|
+
const snapshot = /** @type {{status?: string}} */ (readNodeSnapshot(runDir, id));
|
|
56
|
+
if (SETTLED.has(/** @type {string} */ (snapshot.status))) ids.add(id);
|
|
57
|
+
}
|
|
58
|
+
return ids;
|
|
59
|
+
}
|
|
60
|
+
|
|
31
61
|
/**
|
|
32
62
|
* The bounded `k/n · argv` shape a running node's status surfaces while a
|
|
33
63
|
* verification command is in flight: the command's 1-based position among
|
|
@@ -108,7 +138,7 @@ export async function executeControllerVerification(contract, runDir, node, stat
|
|
|
108
138
|
};
|
|
109
139
|
writeNode(runDir, state, lock);
|
|
110
140
|
const workspace = attemptWorkspace(state) ?? contract.cwd;
|
|
111
|
-
const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node)];
|
|
141
|
+
const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node, settledSiblingIds(runDir, contract, node))];
|
|
112
142
|
/** @param {VerificationAttempt} attempt @returns {VerificationProgress} */
|
|
113
143
|
const progressFor = (attempt) => verificationProgress(attempt.commandIndex + 1, commands.length, /** @type {VerificationCommand|undefined} */ (commands[attempt.commandIndex])?.argv);
|
|
114
144
|
try {
|
|
@@ -221,16 +251,37 @@ function argvEqual(a, b) {
|
|
|
221
251
|
if (!Array.isArray(a) || !Array.isArray(b) || a.length !== b.length) return false;
|
|
222
252
|
return a.every((item, index) => item === b[index]);
|
|
223
253
|
}
|
|
254
|
+
/**
|
|
255
|
+
* Marks whether the integration candidate's own verification pass is running
|
|
256
|
+
* right now, nested inside the attempt's own already-completed verification
|
|
257
|
+
* record rather than as a new node-snapshot field: that record's validator
|
|
258
|
+
* (`validateVerificationSnapshot`) accepts extra keys, where the node
|
|
259
|
+
* snapshot's own strict field list would reject one. Status surfaces read it
|
|
260
|
+
* to tell "still verifying the sealed candidate" apart from "waiting on the
|
|
261
|
+
* judge", which otherwise both read as whatever phase the attempt last
|
|
262
|
+
* dispatched under.
|
|
263
|
+
*
|
|
264
|
+
* @param {string} runDir
|
|
265
|
+
* @param {NodeSnapshot} state
|
|
266
|
+
* @param {LockHandle|null} lock
|
|
267
|
+
* @param {boolean} active
|
|
268
|
+
*/
|
|
269
|
+
function markCandidateVerification(runDir, state, lock, active) {
|
|
270
|
+
state.verification = /** @type {VerificationState} */ ({ ...(state.verification ?? { passed: false, commands: [], completed: false }), candidate: active });
|
|
271
|
+
writeNode(runDir, state, lock);
|
|
272
|
+
}
|
|
224
273
|
/**
|
|
225
274
|
* @param {ValidatedContract} contract
|
|
226
275
|
* @param {ValidatedNode} node
|
|
227
276
|
* @param {NodeSnapshot} state
|
|
228
277
|
* @param {string} runDir
|
|
229
278
|
* @param {string} workspace
|
|
279
|
+
* @param {LockHandle|null} [lock]
|
|
230
280
|
* @returns {Promise<import("../repo/integrate.mjs").CandidateEvidence>}
|
|
231
281
|
*/
|
|
232
|
-
export async function verifyCandidateWorkspace(contract, node, state, runDir, workspace) {
|
|
233
|
-
const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node)];
|
|
282
|
+
export async function verifyCandidateWorkspace(contract, node, state, runDir, workspace, lock = null) {
|
|
283
|
+
const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node, settledSiblingIds(runDir, contract, node))];
|
|
284
|
+
markCandidateVerification(runDir, state, lock, true);
|
|
234
285
|
try {
|
|
235
286
|
const result = await runVerification(commands, workspace, {
|
|
236
287
|
logDir: join(runDir, "logs", `${node.id}.${state.attempt}.candidate-verification`),
|
|
@@ -252,5 +303,7 @@ export async function verifyCandidateWorkspace(contract, node, state, runDir, wo
|
|
|
252
303
|
return settled.retried ? { ...compacted, retried: settled.retried } : compacted;
|
|
253
304
|
} catch (error) {
|
|
254
305
|
return { passed: false, error: boundedUtf8(errorMessage(error), 4 * 1024) };
|
|
306
|
+
} finally {
|
|
307
|
+
markCandidateVerification(runDir, state, lock, false);
|
|
255
308
|
}
|
|
256
309
|
}
|