faberun 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -31,15 +31,15 @@ import {
31
31
  terminalErrorCode,
32
32
  } from "./lifecycle.mjs";
33
33
  import { delay, errorCode } from "../util.mjs";
34
- import { alreadyNotified, notifyQueueFor, notifyQueuesByRun, renderCampaignHandoffSafely } from "./notify-queue.mjs";
35
- import { detectStalls, terminateProcess } from "./process.mjs";
34
+ import { alreadyNotified, emitNodeAdvisories, notifyQueueFor, notifyQueuesByRun, renderCampaignHandoffSafely } from "./notify-queue.mjs";
35
+ import { detectStalls, invocationAlive, terminateProcess } from "./process.mjs";
36
36
  import { transition, writeNode } from "./state.mjs";
37
37
  import { listNodeSnapshots, readNodeSnapshot } from "../run/node-store.mjs";
38
38
  import { render, renderFinalReport, writeFindingsArtifact } from "../report/final.mjs";
39
39
  import { operationNextState, providerReceipts, settleInvocation } from "../run/operations.mjs";
40
40
  import { appendUsageRecord, invocationCost, invocationUsage, recordInvocationUsage } from "../run/usage.mjs";
41
41
  import { captureNodeScopeBoundaries, checkWorkerScope, emptyScope } from "./scope.mjs";
42
- import { validateContract } from "../contract/index.mjs";
42
+ import { validateContractForLaunch } from "../campaign/chain.mjs";
43
43
  import { validateNodeSnapshot } from "../contract/snapshot.mjs";
44
44
  import { finalVerificationCommands, sharedVerificationCommands } from "../contract/final-verification.mjs";
45
45
  import { startJudge, startWorker } from "./dispatch.mjs";
@@ -131,16 +131,22 @@ export function nodeBudgetBasisMs(contract, node) {
131
131
  * never touches phase 2's parked `blocked`/`failed`/`exhausted`/`stalled`
132
132
  * states, and never overwrites the `runtime_tier_exhausted` waiting shape.
133
133
  *
134
+ * A node whose closed job is being settled in the background (its id is a key
135
+ * of `pendingSettlements`) is not this dead end either: its invocation has
136
+ * already exited and left `running`, but the settlement promise still owns
137
+ * deciding what happens to it, so it is left alone until that promise resolves.
138
+ *
134
139
  * @param {string} runDir
135
140
  * @param {Map<string, NodeSnapshot>} states
136
141
  * @param {Map<string, Job>} running
137
142
  * @param {LockHandle|null} lock
143
+ * @param {Map<string, Promise<void>>} [pendingSettlements]
138
144
  * @returns {string[]} the node ids this pass parked
139
145
  */
140
- export function enforceRunningInvariant(runDir, states, running, lock) {
146
+ export function enforceRunningInvariant(runDir, states, running, lock, pendingSettlements = new Map()) {
141
147
  const parked = [];
142
148
  for (const [nodeId, state] of states) {
143
- if (state.status !== "running" || running.has(nodeId)) continue;
149
+ if (state.status !== "running" || running.has(nodeId) || pendingSettlements.has(nodeId)) continue;
144
150
  transition(runDir, state, "blocked", {
145
151
  phase: state.phase,
146
152
  error: {
@@ -175,14 +181,17 @@ function renderFingerprint(states) {
175
181
 
176
182
  /**
177
183
  * @param {string} contractPath
178
- * @param {{detachedBootstrap?: boolean}} [options] `detachedBootstrap` is set
179
- * only by the CLI entry when this process is its own detached child, and
180
- * makes the controller wait for the launcher's acknowledgement
184
+ * @param {{detachedBootstrap?: boolean, baseRef?: string}} [options]
185
+ * `detachedBootstrap` is set only by the CLI entry when this process is its
186
+ * own detached child, and makes the controller wait for the launcher's
187
+ * acknowledgement; `baseRef` is the CLI's own `--base-ref`, re-validated
188
+ * here so a launch and the run it starts agree about what the contract was
189
+ * checked against
181
190
  * @returns {Promise<RunOutcome>}
182
191
  */
183
192
  export async function runContract(contractPath, options = {}) {
184
193
  const absoluteContractPath = resolve(contractPath);
185
- const contract = validateContract(JSON.parse(readFileSync(absoluteContractPath, "utf8")), absoluteContractPath);
194
+ const contract = validateContractForLaunch(JSON.parse(readFileSync(absoluteContractPath, "utf8")), absoluteContractPath, { baseRef: options.baseRef });
186
195
  const runDir = join(contract.cwd, ".runs", contract.id);
187
196
  if (existsSync(runDir)) throw new Error(`run already exists: ${runDir}`);
188
197
  mkdirSync(join(contract.cwd, ".runs"), { recursive: true });
@@ -319,12 +328,13 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
319
328
  statusFingerprint = fingerprint;
320
329
  render(runDir, runsDir, contract, states, renderLock);
321
330
  };
322
- // The tick that owns a running node's verification can be minutes long
323
- // (`executeControllerVerification` is awaited on the critical path below),
324
- // and that whole time status.json would otherwise report whatever the last
325
- // tick left it at. A timer renders between ticks too; it is cheap even when
326
- // idle because `renderFingerprint` still change-detects, so a quiet run
327
- // writes nothing extra.
331
+ // A node's own settlement (its controller verification, its judge round) can
332
+ // be minutes long, and it now runs off the tick's critical path in
333
+ // `pendingSettlements` so dispatch never waits behind it -- but that whole
334
+ // time status.json would otherwise report whatever the last tick left it at.
335
+ // A timer renders between ticks too; it is cheap even when idle because
336
+ // `renderFingerprint` still change-detects, so a quiet run writes nothing
337
+ // extra.
328
338
  const statusTimer = setInterval(() => renderStatusIfChanged(), contract.pollIntervalMs);
329
339
  statusTimer.unref();
330
340
  let handoffFingerprint = statesFingerprint(states);
@@ -364,14 +374,31 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
364
374
 
365
375
  /** @type {Map<string, Job>} */
366
376
  const running = new Map();
377
+ // One promise per node currently settling a closed job -- its controller
378
+ // verification, its candidate verification, its judge round -- kept off the
379
+ // tick's critical path so an eligible sibling still dispatches into a free
380
+ // slot while this node's own invocation has already exited. A node's own
381
+ // steps stay ordered because only one settlement per node is ever in flight
382
+ // (see the dispatch loop below); a *different* node's settlement is queued
383
+ // behind whichever one is already running (`settlementQueue`), not run
384
+ // alongside it -- `finalizeClosedJobs` shares state a concurrent second call
385
+ // would corrupt: the phase-continuation selection that picks at most one
386
+ // live node to carry a session forward, and `repo/integrate.mjs`'s one
387
+ // candidate ref and worktree per run. Only *dispatching a sibling* skips
388
+ // ahead of a node's settlement; two nodes' settlements never interleave.
389
+ /** @type {Map<string, Promise<void>>} */
390
+ const pendingSettlements = new Map();
391
+ /** @type {Promise<void>} */
392
+ let settlementQueue = Promise.resolve();
367
393
  let canceled = false;
368
394
  const cancel = () => { canceled = true; };
369
395
  process.once("SIGINT", cancel);
370
396
  process.once("SIGTERM", cancel);
371
397
  process.once("SIGHUP", cancel);
372
398
  // The heartbeat's `at` is owned by an unref'd timer inside this writer, never
373
- // by the loop body below: the loop awaits controller verification on its own
374
- // critical path and must keep answering "the process is alive" while it does.
399
+ // by the loop body below: a node's settlement can still run for minutes off
400
+ // the tick's critical path (`pendingSettlements`), and the heartbeat must
401
+ // keep answering "the process is alive" while it does.
375
402
  const heartbeat = createHeartbeat({ runDir, intervalMs: HEARTBEAT_INTERVAL_MS });
376
403
  const heartbeatNodes = new Map(contract.nodes.map((node) => [node.id, node]));
377
404
  let heartbeatFingerprint = statesFingerprint(states);
@@ -380,11 +407,89 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
380
407
  const activeHeartbeatNodes = () => [...states.values()]
381
408
  .filter((state) => state.status === "running" && heartbeatNodes.has(state.id))
382
409
  .map((state) => ({ nodeId: state.id, budgetBasis: nodeBudgetBasisMs(contract, /** @type {ValidatedNode} */ (heartbeatNodes.get(state.id))) }));
410
+ // A programmer error surfacing inside a background settlement must still
411
+ // crash the whole run, exactly as an unguarded `await finalizeClosedJobs`
412
+ // used to: it is recorded here and thrown from the top of the loop on the
413
+ // very next tick, rather than immediately, so it cannot itself become the
414
+ // block a sibling's dispatch is waiting behind. A lost lock is not this --
415
+ // the loop's own `lock.assert()` calls surface that same condition on their
416
+ // own schedule, so it is left for them.
417
+ /** @type {unknown} */
418
+ let backgroundSettlementFailure = null;
419
+ // Mark every job that closed this tick as settling, and free its slot,
420
+ // without waiting for any of them: that alone is what lets an eligible
421
+ // sibling dispatch into the freed slot while this node's minutes-long
422
+ // controller verification or judge round is still running. The actual
423
+ // settlement work is chained onto `settlementQueue`, one node at a time in
424
+ // the order its job closed, so it still runs exactly as serialized against
425
+ // every *other* node's settlement as it did when this loop awaited
426
+ // `finalizeClosedJobs` directly -- `finalizeClosedJobs` runs against a
427
+ // one-entry map per node, so this is one call per node rather than the one
428
+ // batched call it used to be, but the chain still runs them one at a time.
429
+ // A settlement may itself dispatch the node's next phase (a judge, a
430
+ // revision) through the same `startJudge`/`startWorker` calls dispatch below
431
+ // uses; those land in `slot`, not the real `running`, so they are copied
432
+ // back into `running` in a `finally` -- unconditionally, win or lose, so a
433
+ // job a settlement started before failing (a lost lock, a programmer error)
434
+ // is still visible to the cleanup sweeps below rather than leaked. The
435
+ // node's own steps stay ordered by never starting a second settlement for a
436
+ // node whose first has not yet cleared `pendingSettlements`.
437
+ const settleClosedJobsInBackground = () => {
438
+ for (const [nodeId, job] of [...running]) {
439
+ if (pendingSettlements.has(nodeId) || !job.closed || invocationAlive(job.invocation)) continue;
440
+ running.delete(nodeId);
441
+ const slot = new Map([[nodeId, job]]);
442
+ const settlement = settlementQueue
443
+ .then(() => finalizeClosedJobs(contract, runDir, states, slot, lock, campaign.path))
444
+ .finally(() => {
445
+ for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
446
+ pendingSettlements.delete(nodeId);
447
+ });
448
+ // The queue itself must never reject -- a rejected settlement (a lost
449
+ // lock, a programmer error) would otherwise wedge every node queued
450
+ // behind it. The rejection still reaches whoever awaits the real
451
+ // `settlement` promise (`pendingSettlements`, below).
452
+ settlementQueue = settlement.catch(() => {});
453
+ // Handled here so an in-flight settlement never becomes an unhandled
454
+ // rejection when nobody happens to await `pendingSettlements` before the
455
+ // process exits; the original promise, still held below, carries the
456
+ // rejection to whichever checkpoint (cancel, shutdown) awaits it.
457
+ settlement.catch((error) => {
458
+ if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
459
+ });
460
+ pendingSettlements.set(nodeId, settlement);
461
+ }
462
+ };
463
+ // Captured once, before the loop, rather than re-derived every tick: it is
464
+ // "parked when this controller invocation started" (autoRetryParkedNodes's
465
+ // own contract), and a node a background settlement parks between two ticks
466
+ // -- rather than synchronously within one, as it always did before
467
+ // settlement moved off the tick's critical path -- must still read as newly
468
+ // parked whichever later tick first observes it, not as already-parked
469
+ // because a per-tick snapshot happened to be taken after it landed.
470
+ const parkedBefore = new Set([...states.values()].filter((state) => PARKED.has(state.status)).map((state) => state.id));
471
+ const anyUnsettled = () => [...states.values()].some((state) => !SETTLED.has(state.status));
383
472
  try {
384
- while ([...states.values()].some((state) => !SETTLED.has(state.status))) {
473
+ // A `while` that re-checked this at the very top of every tick would exit
474
+ // the instant a background settlement flips the run's last unsettled node
475
+ // straight to a terminal status between two ticks -- before the tick body
476
+ // that would have run `autoRetryParkedNodes` against that new status ever
477
+ // gets to. The entry guard skips the loop entirely when there is nothing
478
+ // to do at all (a resume of an already-settled run still does zero
479
+ // iterations); once inside, the exit check moves to the bottom, after the
480
+ // body, so that body always sees a freshly-parked node at least once
481
+ // before the loop is allowed to end.
482
+ if (anyUnsettled()) for (;;) {
385
483
  lock.assert();
484
+ if (backgroundSettlementFailure !== null) throw backgroundSettlementFailure;
386
485
  if (existsSync(join(runDir, "cancel.request.json"))) canceled = true;
387
486
  if (canceled) {
487
+ // A settlement already in flight owns the one decision a cancellation
488
+ // must not race: what this node's own invocation resolved to. Let it
489
+ // land on its real terminal status (and, when it re-dispatched a judge
490
+ // or a revision, on the new job that landed in `running`) before this
491
+ // branch decides which nodes are merely canceled.
492
+ await Promise.all([...pendingSettlements.values()]);
388
493
  const jobs = [...running.values()];
389
494
  await Promise.all(jobs.map((job) => terminateProcess(job)));
390
495
  const envelopes = new Map();
@@ -408,8 +513,14 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
408
513
  break;
409
514
  }
410
515
 
411
- const parkedBefore = new Set([...states.values()].filter((state) => PARKED.has(state.status)).map((state) => state.id));
412
- await finalizeClosedJobs(contract, runDir, states, running, lock, campaign.path);
516
+ // Advisory spend lines are checked every tick, closed job or not: a
517
+ // crossing must be visible while the spend is happening on a node still
518
+ // running, not only once it closes. `finalizeClosedJobs` used to open
519
+ // with this same check; calling it once here, rather than once per
520
+ // node settled this tick, is what keeps it at one pass per tick now
521
+ // that settlement runs per node in `settleClosedJobsInBackground`.
522
+ await emitNodeAdvisories(contract, runDir, states);
523
+ settleClosedJobsInBackground();
413
524
  await detectStalls(contract, running, async (job, status, error) => {
414
525
  const envelope = recordInvocationUsage(job);
415
526
  job.state.usage = invocationUsage(job.state);
@@ -450,8 +561,11 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
450
561
  writeNode(runDir, job.state, lock);
451
562
  });
452
563
  // A node left `running` with no job is a dead end; park it before the
453
- // dispatch pass so it cannot hide behind a healthy sibling.
454
- enforceRunningInvariant(runDir, states, running, lock);
564
+ // dispatch pass so it cannot hide behind a healthy sibling. A node
565
+ // whose closed job is settling in the background is not that dead end
566
+ // -- `pendingSettlements` is what tells this apart from one truly
567
+ // abandoned.
568
+ enforceRunningInvariant(runDir, states, running, lock, pendingSettlements);
455
569
  // Before dependants are blocked, a node that parked on this tick gets
456
570
  // its one automatic retry: it becomes pending, so `blockDependents` sees
457
571
  // nothing to block and the dependants stay `pending`/`phase: "waiting"`
@@ -463,14 +577,37 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
463
577
  if (slots > 0) {
464
578
  const ready = contract.nodes.filter((node) => {
465
579
  const state = states.get(node.id);
466
- return state?.status === "pending" && node.dependsOn.every((id) => states.get(id)?.status === "done");
580
+ return state?.status === "pending" && !pendingSettlements.has(node.id)
581
+ && node.dependsOn.every((id) => states.get(id)?.status === "done");
467
582
  });
468
583
  for (const node of ready.slice(0, slots)) {
469
584
  const state = states.get(node.id);
470
585
  if (!state || routingBackoffActive(state, state.phase)) continue;
586
+ // A node recovered pending a re-ask judge (its own worker attempt
587
+ // already accepted, `state.result` durable) reaches `settleDone` /
588
+ // `integrateAttempt` exactly like a closed job's own settlement
589
+ // does, on the same per-run candidate ref and worktree
590
+ // `settlementQueue` exists to serialize -- so it is dispatched the
591
+ // same way: chained onto the queue rather than awaited here, using
592
+ // its own one-entry `slot` merged back into `running` in a
593
+ // `finally`. The node's own order is untouched (still one entry at
594
+ // a time, gated by `pendingSettlements`); only the tick stops
595
+ // waiting behind it.
471
596
  if (state.phase === "judge" && state.result) {
472
- await applyJudgeRound(await startJudge(contract, node, state, runDir, running, state.result, lock, states, campaign.path),
473
- contract, node, state, runDir, running, lock, states, campaign.path, state.result);
597
+ const workerResult = state.result;
598
+ const slot = new Map();
599
+ const settlement = settlementQueue
600
+ .then(() => startJudge(contract, node, state, runDir, slot, workerResult, lock, states, campaign.path))
601
+ .then((round) => applyJudgeRound(round, contract, node, state, runDir, slot, lock, states, campaign.path, workerResult))
602
+ .finally(() => {
603
+ for (const [settledId, settledJob] of slot) running.set(settledId, settledJob);
604
+ pendingSettlements.delete(node.id);
605
+ });
606
+ settlementQueue = settlement.catch(() => {});
607
+ settlement.catch((error) => {
608
+ if (!(error instanceof LockLostError) && backgroundSettlementFailure === null) backgroundSettlementFailure = error;
609
+ });
610
+ pendingSettlements.set(node.id, settlement);
474
611
  continue;
475
612
  }
476
613
  state.attempt += 1;
@@ -499,10 +636,46 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
499
636
  }
500
637
  }
501
638
  heartbeat.setActive(activeHeartbeatNodes());
502
- if ([...states.values()].some((state) => !SETTLED.has(state.status))) await delay(contract.pollIntervalMs);
639
+ // A background settlement can flip a node straight to a parked
640
+ // terminal status during this same tick's own later awaits
641
+ // (`detectStalls`, `notifyStateChanges`), after the auto-retry pass
642
+ // above already ran and saw it as not-yet-parked. Running it once more
643
+ // here, immediately before the exit check, is what stops the loop from
644
+ // mistaking that freshly-parked node for settled and exiting before it
645
+ // ever got its automatic retry; a node it reopens dispatches on the
646
+ // next tick rather than this one, which `anyUnsettled()` below still
647
+ // correctly keeps the loop alive for.
648
+ autoRetryParkedNodes(contract, runDir, states, lock, parkedBefore);
649
+ if (!anyUnsettled()) break;
650
+ await delay(contract.pollIntervalMs);
503
651
  }
652
+ // The loop above exits (by the bottom check above or the cancel branch's
653
+ // `break`) the instant every node's state looks settled, but a
654
+ // settlement's own trailing work -- sealing a candidate's acceptance,
655
+ // removing its worktree, enqueueing its terminal notification -- can
656
+ // still be running after the transition that made the state look
657
+ // terminal. Nothing past this point (the final render, the findings
658
+ // artifact, the run-terminal notification, releasing the lock) may run
659
+ // ahead of that trailing work.
660
+ await Promise.all([...pendingSettlements.values()]);
661
+ // A settlement's own `finally` deletes it from `pendingSettlements` the
662
+ // instant it settles, win or lose -- so a rejection recorded here can
663
+ // already be gone from the map above by the time this line runs, with
664
+ // nothing left to await it. The loop's own top-of-tick check
665
+ // (`if (backgroundSettlementFailure !== null) throw ...`) cannot save
666
+ // that case either: the loop has already exited. Checked again here,
667
+ // once, so a background settlement failure can never read as a clean
668
+ // finish just because it settled on the same tick the run's last node
669
+ // did.
670
+ if (backgroundSettlementFailure !== null) throw backgroundSettlementFailure;
504
671
  } catch (error) {
505
672
  if (!(error instanceof LockLostError)) throw error;
673
+ // A settlement still merges whatever it started into `running` in its own
674
+ // `finally`, win or lose, regardless of whether anything awaits it; wait
675
+ // for that to land (never rejecting itself, so a second lock-loss here
676
+ // cannot mask the one already being handled) before the termination sweep
677
+ // reads `running`.
678
+ await Promise.allSettled([...pendingSettlements.values()]);
506
679
  await Promise.all([...running.values()].map((job) => terminateProcess(job)));
507
680
  notifyQueuesByRun.delete(runDir);
508
681
  return { runDir, states, ok: false, error };
@@ -0,0 +1,141 @@
1
+ /**
2
+ * The judge branch of `finalizeClosedJobs`, and the workspace comparison its
3
+ * fail-closed write check depends on. Split out of `lifecycle.mjs` for the
4
+ * same reason `engine/settle.mjs` was: `lifecycle.mjs` and
5
+ * `test/engine/judge.test.mjs` both sit on the 800-line ceiling
6
+ * `test/repo/source-shape.test.mjs` enforces.
7
+ *
8
+ * `settleJudgeRound` runs only after `lifecycle.mjs` has already called
9
+ * `judgeWorkspaceWriteViolation` and found nothing: the check has to run
10
+ * before any branch here (a failed provider, a bounded re-dispatch, or a
11
+ * verdict) can adopt or launder a judge's own write, so it cannot live inside
12
+ * this function without reintroducing the escape it closes.
13
+ *
14
+ * `clearTierExhaustion` and `handleProviderExhaustion` are `lifecycle.mjs`'s
15
+ * own -- importing them back from there would recreate the exact
16
+ * `lifecycle.mjs` <-> `review.mjs` cycle `test/repo/source-shape.test.mjs`
17
+ * once caught and `AGENTS.md` records as fixed, so the caller passes them in
18
+ * instead.
19
+ */
20
+ import { judgeReaskOutstanding } from "./judge-gate.mjs";
21
+ import { judgeVerdictEvidence } from "../contract/review-modes.mjs";
22
+ import {
23
+ JUDGE_MAX_FAILURES,
24
+ applyJudgeProtocolFailure,
25
+ applyJudgeResult,
26
+ applyJudgeRound,
27
+ settleUnavailableJudge,
28
+ } from "./review.mjs";
29
+ import { networkTransition } from "./backoff.mjs";
30
+ import { startJudge } from "./dispatch.mjs";
31
+ import { writeNode } from "./state.mjs";
32
+ import { compareWorkspaceSnapshot } from "../repo/workspace.mjs";
33
+ import { errorCode, errorMessage, excerpt } from "../util.mjs";
34
+
35
+ /** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
36
+ /** @typedef {import("../contract/index.mjs").NodeSnapshot} NodeSnapshot */
37
+ /** @typedef {import("./process.mjs").Job} Job */
38
+ /** @typedef {import("../run/lock.mjs").LockRecord} LockRecord */
39
+ /** @typedef {ReturnType<typeof import("../run/lock.mjs").acquire>} LockHandle */
40
+ /** @typedef {import("../harnesses/index.mjs").ProviderEnvelope} ProviderEnvelope */
41
+ /** @typedef {ProviderEnvelope & {costProvenance?: "priced"}} PricedEnvelope */
42
+
43
+ /**
44
+ * Whether a closed judge job wrote into its own workspace, or a workspace
45
+ * comparison error that means the same thing: either is a violation, never
46
+ * silence. `compareWorkspaceSnapshot` throws `snapshot_ignore_changed` when
47
+ * the judge edits an ignore source, `snapshot_symlink_escape` when it creates
48
+ * a symlink out of the tree, and `snapshot_too_large` when it adds enough
49
+ * entries -- all ordinary judge writes that must fail closed exactly like the
50
+ * worker path (`engine/scope.mjs`'s `checkWorkerScope`), never vanish into a
51
+ * "no writes" result. `job.scopeBaseline` is null only when the capture at
52
+ * dispatch time itself failed -- `startJudge` degrades to unchecked rather
53
+ * than refuse to run the judge at all -- and that is the one case with
54
+ * nothing to compare, not a violation.
55
+ *
56
+ * @param {Job} job
57
+ * @returns {{message: string}|null}
58
+ */
59
+ export function judgeWorkspaceWriteViolation(job) {
60
+ const baseline = job.scopeBaseline;
61
+ if (!baseline) return null;
62
+ try {
63
+ const { unexpectedPaths } = compareWorkspaceSnapshot(/** @type {import("../repo/workspace.mjs").WorkspaceSnapshot} */ (baseline), job.cwd);
64
+ if (!unexpectedPaths.length) return null;
65
+ return { message: excerpt(`judge wrote into its own workspace (${unexpectedPaths.length}): ${unexpectedPaths.slice(0, 8).join(", ")}`) ?? "judge wrote into its own workspace" };
66
+ } catch (error) {
67
+ return { message: excerpt(`judge workspace comparison failed (${errorCode(error) ?? "scope_snapshot_invalid"}): ${errorMessage(error)}`) ?? "judge workspace comparison failed" };
68
+ }
69
+ }
70
+
71
+ /**
72
+ * Settle a closed judge invocation whose write check already passed: a
73
+ * provider that failed outright gets one bounded re-dispatch, then review-mode
74
+ * settlement; anything else that is not exactly one usable verdict (none at
75
+ * all, several of them, an unparseable one, a stream cut off before its
76
+ * terminal envelope, or a phase killed on its wall clock) takes the same
77
+ * bounded re-ask; a clean verdict is applied.
78
+ *
79
+ * @param {ValidatedContract} contract
80
+ * @param {Job} job
81
+ * @param {NodeSnapshot} state
82
+ * @param {string} runDir
83
+ * @param {Map<string, Job>} running
84
+ * @param {LockHandle} lock
85
+ * @param {Map<string, NodeSnapshot>} states
86
+ * @param {string} campaignPath
87
+ * @param {PricedEnvelope} envelope
88
+ * @param {{clearTierExhaustion: (state: NodeSnapshot) => void, handleProviderExhaustion: typeof import("./lifecycle.mjs").handleProviderExhaustion}} hooks
89
+ * `lifecycle.mjs`'s own tier-exhaustion clear and provider-exhaustion router, passed in rather than imported back.
90
+ * @returns {Promise<void>}
91
+ */
92
+ export async function settleJudgeRound(contract, job, state, runDir, running, lock, states, campaignPath, envelope, { clearTierExhaustion, handleProviderExhaustion }) {
93
+ const node = job.node;
94
+ // A judge provider that failed outright (its turn died, its tool host was
95
+ // gone) is a provider failure, never a verdict: the gate cannot adopt a
96
+ // result the judge could not ground in inspection. Re-dispatch the judge
97
+ // once on the same routing, then settle by review mode so a judge failure
98
+ // is surfaced, never silently settled. A stream that never reached its
99
+ // terminal envelope is a protocol defect instead and takes the bounded
100
+ // re-ask below.
101
+ if (envelope.status === "failed" && envelope.error?.code !== "incomplete_stream") {
102
+ // A judge that lost its socket is not an unavailable judge. It buys the
103
+ // same bounded network waits a worker does, on the runtime it already
104
+ // warmed, and spends none of the one re-dispatch counted below.
105
+ const network = networkTransition(contract, node, state, "judge", envelope, job.exitCode);
106
+ if (network && handleProviderExhaustion(contract, runDir, node, state, "judge", envelope, job.runtime.id, lock, states, campaignPath, network)) return;
107
+ clearTierExhaustion(state);
108
+ // The provider died on the bounded re-ask itself, so the one permitted
109
+ // re-ask is spent: settle by review mode here rather than dispatch a
110
+ // third judge invocation behind a fresh failure count.
111
+ if (judgeReaskOutstanding(state)) {
112
+ await applyJudgeProtocolFailure(contract, node, state, runDir, running, lock, states, campaignPath, envelope.error?.message ?? "judge provider failed");
113
+ return;
114
+ }
115
+ state.judgeFailures = (state.judgeFailures ?? 0) + 1;
116
+ if (state.judgeFailures < JUDGE_MAX_FAILURES) {
117
+ writeNode(runDir, state, lock);
118
+ await applyJudgeRound(await startJudge(contract, node, state, runDir, running, state.result, lock, states, campaignPath),
119
+ contract, node, state, runDir, running, lock, states, campaignPath, state.result);
120
+ return;
121
+ }
122
+ await settleUnavailableJudge(contract, node, state, runDir, lock, states, campaignPath, envelope.error?.message ?? "judge provider failed");
123
+ return;
124
+ }
125
+ // Whatever else this invocation produced, it is not exactly one usable
126
+ // verdict: no verdict at all, several of them in separate agent messages,
127
+ // an unparseable one, a stream cut off before its terminal envelope, or a
128
+ // phase killed on its wall clock. One bounded re-ask, then the review mode
129
+ // decides — advisory completes, blocking enters attention with the work
130
+ // preserved so a retry in place can re-judge it.
131
+ const evidence = judgeVerdictEvidence(envelope);
132
+ if (!evidence.ok) {
133
+ const network = networkTransition(contract, node, state, "judge", envelope, job.exitCode);
134
+ if (network && handleProviderExhaustion(contract, runDir, node, state, "judge", envelope, job.runtime.id, lock, states, campaignPath, network)) return;
135
+ clearTierExhaustion(state);
136
+ await applyJudgeProtocolFailure(contract, node, state, runDir, running, lock, states, campaignPath, evidence.reason);
137
+ return;
138
+ }
139
+ clearTierExhaustion(state);
140
+ await applyJudgeResult(contract, node, state, evidence.result, runDir, lock, running, states, campaignPath);
141
+ }
@@ -165,7 +165,7 @@ export async function settleDone(contract, node, state, runDir, lock, states, ca
165
165
  attemptSha: sealed.sha,
166
166
  branch: state.worktree.branch,
167
167
  verificationEvidence: state.verification,
168
- verifyCandidate: (candidateWorkspace) => verifyCandidateWorkspace(contract, node, state, runDir, candidateWorkspace),
168
+ verifyCandidate: (candidateWorkspace) => verifyCandidateWorkspace(contract, node, state, runDir, candidateWorkspace, lock),
169
169
  onAccepted: async (transaction) => {
170
170
  const acceptedPath = state.worktree?.path ?? attemptWorktreePath(runDir, contract.id, node.id, transaction.attempt);
171
171
  if (state.attempt === transaction.attempt && state.status !== "done") {
@@ -11,8 +11,10 @@
11
11
  import { attemptWorkspace } from "../repo/worktree.mjs";
12
12
  import { boundedUtf8, errorMessage } from "../util.mjs";
13
13
  import { compactVerification } from "../contract/verification.mjs";
14
- import { finalVerificationCommands, sharedVerificationCommands } from "../contract/final-verification.mjs";
14
+ import { finalVerificationCommands, phaseTerminalNode, sharedVerificationCommands } from "../contract/final-verification.mjs";
15
15
  import { join } from "node:path";
16
+ import { listNodeSnapshots, readNodeSnapshot } from "../run/node-store.mjs";
17
+ import { SETTLED } from "./prompts.mjs";
16
18
  import { terminateInvocation } from "./process.mjs";
17
19
  import { writeNode } from "./state.mjs";
18
20
  import { runVerification } from "./run-command.mjs";
@@ -28,6 +30,34 @@ import { runVerification } from "./run-command.mjs";
28
30
  /** @typedef {import("../contract/index.mjs").VerificationState} VerificationState */
29
31
  /** @typedef {{index: number, total: number, argv: string}} VerificationProgress */
30
32
 
33
+ /**
34
+ * The sibling phase-terminal node ids already settled, read straight off
35
+ * disk: every node snapshot is persisted from the run's first tick (see
36
+ * `runContract`), so a node still `pending` reads back honestly, not as
37
+ * missing. `finalVerificationCommands` uses this to decide, at the moment a
38
+ * phase-terminal node's own verification runs, whether it is the one that
39
+ * closes the phase -- recomputed fresh on every call, so a retried node sees
40
+ * its siblings' current status each time, not a decision frozen from an
41
+ * earlier attempt.
42
+ *
43
+ * @param {string} runDir
44
+ * @param {ValidatedContract} contract
45
+ * @param {ValidatedNode} node
46
+ * @returns {Set<string>}
47
+ */
48
+ function settledSiblingIds(runDir, contract, node) {
49
+ const ids = new Set();
50
+ for (const name of listNodeSnapshots(runDir)) {
51
+ const id = name.slice(0, -".json".length);
52
+ if (id === node.id) continue;
53
+ const sibling = contract.nodes.find((candidate) => candidate.id === id);
54
+ if (!sibling || !phaseTerminalNode(contract, sibling)) continue;
55
+ const snapshot = /** @type {{status?: string}} */ (readNodeSnapshot(runDir, id));
56
+ if (SETTLED.has(/** @type {string} */ (snapshot.status))) ids.add(id);
57
+ }
58
+ return ids;
59
+ }
60
+
31
61
  /**
32
62
  * The bounded `k/n · argv` shape a running node's status surfaces while a
33
63
  * verification command is in flight: the command's 1-based position among
@@ -108,7 +138,7 @@ export async function executeControllerVerification(contract, runDir, node, stat
108
138
  };
109
139
  writeNode(runDir, state, lock);
110
140
  const workspace = attemptWorkspace(state) ?? contract.cwd;
111
- const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node)];
141
+ const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node, settledSiblingIds(runDir, contract, node))];
112
142
  /** @param {VerificationAttempt} attempt @returns {VerificationProgress} */
113
143
  const progressFor = (attempt) => verificationProgress(attempt.commandIndex + 1, commands.length, /** @type {VerificationCommand|undefined} */ (commands[attempt.commandIndex])?.argv);
114
144
  try {
@@ -221,16 +251,37 @@ function argvEqual(a, b) {
221
251
  if (!Array.isArray(a) || !Array.isArray(b) || a.length !== b.length) return false;
222
252
  return a.every((item, index) => item === b[index]);
223
253
  }
254
+ /**
255
+ * Marks whether the integration candidate's own verification pass is running
256
+ * right now, nested inside the attempt's own already-completed verification
257
+ * record rather than as a new node-snapshot field: that record's validator
258
+ * (`validateVerificationSnapshot`) accepts extra keys, where the node
259
+ * snapshot's own strict field list would reject one. Status surfaces read it
260
+ * to tell "still verifying the sealed candidate" apart from "waiting on the
261
+ * judge", which otherwise both read as whatever phase the attempt last
262
+ * dispatched under.
263
+ *
264
+ * @param {string} runDir
265
+ * @param {NodeSnapshot} state
266
+ * @param {LockHandle|null} lock
267
+ * @param {boolean} active
268
+ */
269
+ function markCandidateVerification(runDir, state, lock, active) {
270
+ state.verification = /** @type {VerificationState} */ ({ ...(state.verification ?? { passed: false, commands: [], completed: false }), candidate: active });
271
+ writeNode(runDir, state, lock);
272
+ }
224
273
  /**
225
274
  * @param {ValidatedContract} contract
226
275
  * @param {ValidatedNode} node
227
276
  * @param {NodeSnapshot} state
228
277
  * @param {string} runDir
229
278
  * @param {string} workspace
279
+ * @param {LockHandle|null} [lock]
230
280
  * @returns {Promise<import("../repo/integrate.mjs").CandidateEvidence>}
231
281
  */
232
- export async function verifyCandidateWorkspace(contract, node, state, runDir, workspace) {
233
- const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node)];
282
+ export async function verifyCandidateWorkspace(contract, node, state, runDir, workspace, lock = null) {
283
+ const commands = [...node.taskPacket.verification, ...sharedVerificationCommands(contract), ...finalVerificationCommands(contract, node, settledSiblingIds(runDir, contract, node))];
284
+ markCandidateVerification(runDir, state, lock, true);
234
285
  try {
235
286
  const result = await runVerification(commands, workspace, {
236
287
  logDir: join(runDir, "logs", `${node.id}.${state.attempt}.candidate-verification`),
@@ -252,5 +303,7 @@ export async function verifyCandidateWorkspace(contract, node, state, runDir, wo
252
303
  return settled.retried ? { ...compacted, retried: settled.retried } : compacted;
253
304
  } catch (error) {
254
305
  return { passed: false, error: boundedUtf8(errorMessage(error), 4 * 1024) };
306
+ } finally {
307
+ markCandidateVerification(runDir, state, lock, false);
255
308
  }
256
309
  }