faberun 0.18.0 → 0.19.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/integrations/claude-code/statusline.sh +12 -2
  2. package/package.json +2 -2
  3. package/skills/init-agentkit/scripts/install-agentkit.sh +10 -2
  4. package/src/campaign/chain.mjs +5 -4
  5. package/src/cli/brand.mjs +4 -5
  6. package/src/cli/launch.mjs +11 -2
  7. package/src/cli.mjs +10 -4
  8. package/src/contract/index.mjs +16 -3
  9. package/src/contract/task-packet.mjs +13 -1
  10. package/src/engine/bulk-read.mjs +7 -1
  11. package/src/engine/cancel.mjs +3 -3
  12. package/src/engine/failover.mjs +11 -4
  13. package/src/engine/gate.mjs +17 -3
  14. package/src/engine/judge-gate.mjs +9 -13
  15. package/src/engine/live-preflight.mjs +43 -10
  16. package/src/engine/live-silence.mjs +49 -0
  17. package/src/engine/process-identity.mjs +12 -1
  18. package/src/engine/process.mjs +22 -3
  19. package/src/engine/resume.mjs +2 -1
  20. package/src/engine/run-command.mjs +8 -5
  21. package/src/engine/run-identity.mjs +213 -14
  22. package/src/engine/runtime-discovery.mjs +37 -5
  23. package/src/engine/scheduler.mjs +14 -23
  24. package/src/engine/scope.mjs +13 -6
  25. package/src/engine/supervise.mjs +2 -1
  26. package/src/harnesses/catalogue.mjs +8 -1
  27. package/src/harnesses/dsh/runner.mjs +6 -1
  28. package/src/harnesses/index.mjs +35 -7
  29. package/src/host/home.mjs +38 -0
  30. package/src/host/platform.mjs +204 -1
  31. package/src/host/preflight.mjs +183 -19
  32. package/src/notify/index.mjs +7 -1
  33. package/src/notify/session.mjs +6 -1
  34. package/src/plan/pipeline.mjs +14 -4
  35. package/src/plan/preflight.mjs +77 -0
  36. package/src/plan/template.mjs +46 -14
  37. package/src/repo/declared-paths.mjs +87 -6
  38. package/src/repo/signal.mjs +56 -21
  39. package/src/repo/workspace.mjs +3 -2
  40. package/src/repo/worktree.mjs +2 -1
  41. package/src/report/message.mjs +7 -7
  42. package/src/report/next.mjs +15 -1
  43. package/src/run/availability.mjs +138 -0
  44. package/src/run/disk-gc.mjs +16 -2
  45. package/src/run/lock.mjs +8 -0
  46. package/src/run/node-store.mjs +33 -0
  47. package/src/run/paths.mjs +13 -0
  48. package/src/seat/allowance.mjs +4 -1
  49. package/src/seat/tmux.mjs +9 -1
  50. package/src/util.mjs +46 -0
@@ -0,0 +1,49 @@
1
+ /**
2
+ * What counts as a provider saying nothing at all.
3
+ *
4
+ * This is one rule with three readers -- the dispatch gate
5
+ * (`engine/run-identity.mjs`), planning's pre-stage refusal
6
+ * (`plan/preflight.mjs`) and `doctor` (`host/preflight.mjs`) -- and it is its
7
+ * own module because those three cannot all import each other. `doctor` owns
8
+ * `reachableRuntimes`, which `engine/live-preflight.mjs` imports, so a
9
+ * `doctor` that read the rule out of the engine closed a runtime import
10
+ * cycle. This module imports nothing: it reads a probe's recorded detail and
11
+ * says whether a provider answered, and nothing else.
12
+ *
13
+ * The rule itself is the campaign's claim in one line. Any verdict a provider
14
+ * returned is an answer, a quota refusal included, and the run proceeds onto
15
+ * whatever the contract declares. Only silence blocks.
16
+ */
17
+
18
+ /**
19
+ * The causes that mean no runtime said anything at all: the provider was
20
+ * asked and did not answer, or could not be started to be asked.
21
+ *
22
+ * `command_invalid` is deliberately not one of them. A command that could not
23
+ * be constructed never reached a provider, so there is no availability
24
+ * verdict either way -- that is a contract defect and validation already owns
25
+ * it. Measured 2026-09-22: three deterministic evals declare a fallback and a
26
+ * judge runtime they never invoke, so those carry no replay recording and the
27
+ * replay adapter throws when asked to build their command. Blocking there
28
+ * refuses a run over a runtime it would never have used, for a fault the
29
+ * provider never had.
30
+ */
31
+ const LIVE_SILENCE_CAUSES = new Set(["preflight_timeout", "spawn_error"]);
32
+
33
+ /**
34
+ * Whether a live probe is pipeline silence rather than a verdict, and which
35
+ * cause. `preflightContract` embeds the provider envelope's error code in the
36
+ * probe detail (`… · live failed · <code>: …`), and the repository-failure
37
+ * wording means no runtime was even asked. Everything else -- a quota
38
+ * refusal, an auth failure, unparsable output -- is a provider that answered,
39
+ * and an answer is hello enough.
40
+ *
41
+ * @param {import("../harnesses/index.mjs").ProbeResult} probe
42
+ * @returns {string|null}
43
+ */
44
+ export function liveSilenceCause(probe) {
45
+ if (probe.ok) return null;
46
+ if (/live preflight repository failed/u.test(probe.detail ?? "")) return "spawn_error";
47
+ const match = / · live \S+ · ([a-z_]+):/u.exec(probe.detail ?? "");
48
+ return match !== null && LIVE_SILENCE_CAUSES.has(match[1]) ? match[1] : null;
49
+ }
@@ -86,6 +86,17 @@ export function processStartTokenMatches(invocation) {
86
86
  * -- a pid this user cannot signal is not the child this controller spawned --
87
87
  * and a null token with no handle is unverifiable, so it is never owned.
88
88
  *
89
+ * Except on Windows, which records no token at all: `wmic` is gone from
90
+ * Windows 11 26200 and the PowerShell that replaced it costs about 400 ms a
91
+ * probe (`run/lock.mjs`). Holding POSIX's answer there means every recorded
92
+ * invocation is unverifiable, so a controller never terminates the provider it
93
+ * started: measured 2026-09-20, a suite run left 105 node fixtures alive and
94
+ * then waited on one of them, and `cancel` reported providers it had not
95
+ * stopped. A live pid this controller recorded is the evidence that platform
96
+ * has, and it is the same evidence `killTarget` already acts on there. What is
97
+ * given up is the pid-reuse defence: a pid recycled into an unrelated process
98
+ * between the record and the kill reads as owned.
99
+ *
89
100
  * @param {InvocationProbe} invocation
90
101
  * @param {{child?: ChildProcess|null}} [options]
91
102
  * @returns {boolean}
@@ -106,6 +117,6 @@ export function invocationOwned(invocation, options = {}) {
106
117
  const child = options.child;
107
118
  if (child && child.exitCode === null && child.signalCode === null) return true;
108
119
  const token = invocation.processStartToken;
109
- if (!token) return false;
120
+ if (!token) return process.platform === "win32";
110
121
  return processStartToken(invocation.pid) === token;
111
122
  }
@@ -23,6 +23,7 @@ import { randomUUID } from "node:crypto";
23
23
  import { spawn } from "node:child_process";
24
24
  import { appendJsonl, writeJsonAtomic } from "../run/store.mjs";
25
25
  import { writeNodeSnapshot } from "../run/node-store.mjs";
26
+ import { killTarget } from "../host/platform.mjs";
26
27
 
27
28
  /** @typedef {import("../contract/index.mjs").ValidatedContract} ValidatedContract */
28
29
  /** @typedef {import("../contract/index.mjs").ValidatedNode} ValidatedNode */
@@ -35,7 +36,7 @@ import { writeNodeSnapshot } from "../run/node-store.mjs";
35
36
  /** @typedef {{prompt: string|null, stdout: string, stderr: string}} PathSet */
36
37
  /** @typedef {{id: string, pid: number, processGroupId: number|null, processStartToken: string|null, harness: string, runtimeId: string|null, runtimeFingerprint?: string, revision?: number, phase: string, promptPath: string|null, stdoutPath: string, stderrPath: string, startedAt: string, deadlineAt: string|null, updatedAt: string, closedAt: string|null, exitCode: number|null, signal: string|null, status: "active"|"closed"|"terminated", executable: string, snapshotPath?: string, usage?: Usage, usageEstimated?: boolean, costUsd?: number|null, costProvenance?: "priced", runId?: string, campaignId?: string, nodeId?: string, attempt?: number, workspace?: string, worktreeBranch?: string|null, worktreeBaseSha?: string|null, planPhase?: string, role?: "worker"|"judge", model?: string, reasoning?: string|null, sandbox?: string|null, continuationId?: string|null, continuationMode?: "fresh"|"reuse"|"rotate", session?: import("../harnesses/session-metrics.mjs").SessionLedger|null}} Invocation */
37
38
  /** @typedef {{pid: number|null, processGroupId?: number|null, processStartToken?: string|null}} InvocationProbe */
38
- /** @typedef {{child: ChildProcess, contract: ValidatedContract, node: ValidatedNode, state: NodeSnapshot, runtime: HarnessRuntime & {id: string|null}, cwd: string, paths: PathSet, phase: string, invocation: Invocation, startedAt: string, startedTicks: bigint, progressTicks: bigint, lastOutputAt: number, closed: boolean, exitCode: number|null, signal: string|null, spawnError: Error|null, terminating: Promise<void>|null, gateConfigPath: string, gateReleasePath: string, scopeBaseline?: unknown, scopeChecked?: boolean, scopeViolation?: boolean, resultMaterialization?: boolean, recoveryBaseline?: unknown, observeTimer?: ReturnType<typeof setInterval>, monitorOffset?: number, monitorParser?: import("../harnesses/session-metrics.mjs").SessionMetricsParser, lastEventCount?: number, observedOnce?: boolean, onClose?: (invocation: Invocation) => void, onInvocationUpdate?: (invocation: Invocation) => void, onProgress?: (state: NodeSnapshot) => void}} Job */
39
+ /** @typedef {{child: ChildProcess, contract: ValidatedContract, node: ValidatedNode, state: NodeSnapshot, runtime: HarnessRuntime & {id: string|null}, cwd: string, paths: PathSet, phase: string, invocation: Invocation, startedAt: string, startedTicks: bigint, progressTicks: bigint, lastOutputAt: number, closed: boolean, exitCode: number|null, signal: string|null, spawnError: Error|null, terminating: Promise<void>|null, gateConfigPath: string, gateReleasePath: string, scopeBaseline?: unknown, scopeChecked?: boolean, scopeViolation?: boolean, resultMaterialization?: boolean, recoveryBaseline?: unknown, observeTimer?: ReturnType<typeof setInterval>, monitorOffset?: number, monitorParser?: import("../harnesses/session-metrics.mjs").SessionMetricsParser, lastEventCount?: number, lastMonitorOffset?: number, observedOnce?: boolean, onClose?: (invocation: Invocation) => void, onInvocationUpdate?: (invocation: Invocation) => void, onProgress?: (state: NodeSnapshot) => void}} Job */
39
40
  /** @typedef {{graceMs?: number, killGraceMs?: number, escalate?: boolean, runDir?: string, kill?: (pid: number, signal: string|number) => unknown, child?: ChildProcess|null}} TerminateOptions */
40
41
 
41
42
  const HERE = dirname(fileURLToPath(import.meta.url));
@@ -433,8 +434,26 @@ export async function detectStalls(contract, running, onTimeout, onProgress, onB
433
434
  if (streaming) {
434
435
  const monitored = monitorInvocation(job);
435
436
  const events = monitored.turns + monitored.toolCalls;
436
- if (events !== job.lastEventCount || job.observedOnce !== true) {
437
+ // Transcript bytes the monitor consumed count as liveness beside the
438
+ // events, because a turn can transmit for a long time without finishing
439
+ // one. Codex meters progress as `turn.completed` plus tool calls, so a
440
+ // single long reasoning stretch -- streaming `item.completed` records
441
+ // that are neither -- advanced no counter and was killed as a stall
442
+ // while it was actively transmitting. This is not the mtime the module
443
+ // header rejects: mtime moves for a buffered harness that has written
444
+ // nothing a provider produced, and this branch is the streaming
445
+ // harnesses only, where new bytes on the transcript are the provider's
446
+ // own output and the monitor's offset only ever advances. A process
447
+ // that transmits forever without ending is still held by the wall clock
448
+ // and the turn cap below.
449
+ // An offset this job has never recorded is not growth: `observedOnce`
450
+ // below is what covers the first pass, and reading `undefined` as a
451
+ // change would make every first observation look like progress.
452
+ const consumed = job.monitorOffset ?? 0;
453
+ const grew = consumed !== (job.lastMonitorOffset ?? consumed);
454
+ if (events !== job.lastEventCount || grew || job.observedOnce !== true) {
437
455
  job.lastEventCount = events;
456
+ job.lastMonitorOffset = consumed;
438
457
  job.progressTicks = now;
439
458
  // `lastOutputAt` is the supervised controller's provider-progress
440
459
  // signal (scheduler.mjs): keep it advancing for an event that counts
@@ -572,7 +591,7 @@ function signalInvocation(invocation, signal, options = {}) {
572
591
  const pid = invocation.pid;
573
592
  if (pid === null || pid === undefined) return false;
574
593
  const target = process.platform === "win32" ? pid : -(invocation.processGroupId ?? pid);
575
- const kill = options.kill ?? process.kill;
594
+ const kill = options.kill ?? killTarget;
576
595
  try {
577
596
  kill(target, signal);
578
597
  return true;
@@ -23,7 +23,8 @@ import { attemptWorktreePath } from "../run/paths.mjs";
23
23
  import { canonicalWorkerResultText, isResultMaterializationInvocation, materializeAttemptResult, recoverWorkerResult } from "./result-file.mjs";
24
24
  import { checkPersistedWorkerScope, persistedScopeBoundary, reconcileAmbiguousWorkerRestart, resolveUnknownEffect } from "./scope.mjs";
25
25
  import { closePersistedInvocation, recoverOrphan, recoveryFromOverride } from "./recover.mjs";
26
- import { driveRun, readRunNodes } from "./scheduler.mjs";
26
+ import { driveRun } from "./scheduler.mjs";
27
+ import { readRunNodes } from "../run/node-store.mjs";
27
28
  import { emptyUsage, invocationCost, invocationUsage, persistRecoveryUsage } from "../run/usage.mjs";
28
29
  import { ensureTerminalEvent, hasDoneEvent, recordExecutionOverride, transition, writeNode } from "./state.mjs";
29
30
  import { errorMessage, excerpt } from "../util.mjs";
@@ -17,6 +17,7 @@ import { processStartToken } from "../run/lock.mjs";
17
17
  import { randomUUID } from "node:crypto";
18
18
  import { runMutation } from "./mutation.mjs";
19
19
  import { spawn } from "node:child_process";
20
+ import { killTarget, spawnInvocation } from "../host/platform.mjs";
20
21
  /** @typedef {import("../contract/verification.mjs").VerificationOptions} VerificationOptions */
21
22
 
22
23
  /** @typedef {import("node:child_process").ChildProcess} ChildProcess */
@@ -228,7 +229,11 @@ function runCommand(command, baseCwd, commandCwd, attempt, signal, options, comm
228
229
  };
229
230
  options?.onAttemptStart?.({ ...identity });
230
231
  const env = verificationEnv(command);
231
- child = spawn(command.argv[0], command.argv.slice(1), { cwd, env, detached: process.platform !== "win32", stdio: ["ignore", "pipe", "pipe"] });
232
+ // A verification command names a binary the same way a harness runtime
233
+ // does, and on Windows `npm test` is `npm.cmd`: the invocation, not the
234
+ // raw argv, is what can actually be spawned there.
235
+ const invocation = spawnInvocation(command.argv[0], command.argv.slice(1), { cwd });
236
+ child = spawn(invocation.command, invocation.args, { cwd, env, detached: process.platform !== "win32", stdio: ["ignore", "pipe", "pipe"], ...invocation.options });
232
237
  const pid = child.pid ?? null;
233
238
  let paused = false;
234
239
  if (process.platform !== "win32" && pid) {
@@ -268,8 +273,7 @@ function runCommand(command, baseCwd, commandCwd, attempt, signal, options, comm
268
273
  */
269
274
  function terminateGroup(child) {
270
275
  try {
271
- if (process.platform !== "win32") process.kill(-/** @type {number} */ (child.pid), "SIGTERM");
272
- else child.kill("SIGTERM");
276
+ killTarget(process.platform === "win32" ? /** @type {number} */ (child.pid) : -/** @type {number} */ (child.pid), "SIGTERM");
273
277
  } catch {
274
278
  try { child.kill("SIGTERM"); } catch {
275
279
  // ESRCH: the group kill failed and the leader was already gone.
@@ -277,8 +281,7 @@ function terminateGroup(child) {
277
281
  }
278
282
  setTimeout(() => {
279
283
  try {
280
- if (process.platform !== "win32") process.kill(-/** @type {number} */ (child.pid), "SIGKILL");
281
- else child.kill("SIGKILL");
284
+ killTarget(process.platform === "win32" ? /** @type {number} */ (child.pid) : -/** @type {number} */ (child.pid), "SIGKILL");
282
285
  } catch {
283
286
  try { child.kill("SIGKILL"); } catch {
284
287
  // ESRCH: the SIGKILL fallback found no leader left to kill.
@@ -11,8 +11,9 @@
11
11
  * version on its first call, and a missing version is indistinguishable from a
12
12
  * changed one.
13
13
  */
14
- import { CONTRACT_VERSION, PROTOCOL_SCHEMA_VERSION, probeRuntime } from "../harnesses/index.mjs";
14
+ import { CONTRACT_VERSION, PROTOCOL_SCHEMA_VERSION, getHarness, harnessCapabilities, probeRuntime } from "../harnesses/index.mjs";
15
15
  import { appendJsonl, writeJsonAtomic } from "../run/store.mjs";
16
+ import { availabilityKey, readAvailability, recordAvailability } from "../run/availability.mjs";
16
17
  import { blockingChecks, environmentPreflight, reachableRuntimes } from "../host/preflight.mjs";
17
18
  import { captureSourceIdentity } from "../repo/source-identity.mjs";
18
19
  import { boundedGitSync } from "../repo/worktree.mjs";
@@ -24,8 +25,12 @@ import { stableJson } from "../util.mjs";
24
25
  import { validateRunMetadata } from "../contract/snapshot.mjs";
25
26
  import { contractDigest } from "../contract/index.mjs";
26
27
  import { RUNS_DIR_NAME, runDirectory } from "../run/paths.mjs";
28
+ import { preflightContract } from "./live-preflight.mjs";
29
+ import { liveSilenceCause } from "./live-silence.mjs";
27
30
 
28
31
  /** @typedef {import("../harnesses/index.mjs").HarnessRuntime} HarnessRuntime */
32
+ /** @typedef {import("../harnesses/index.mjs").ProbeResult} ProbeResult */
33
+ /** @typedef {import("../engine/runtime-discovery.mjs").RuntimeAvailability} RuntimeAvailability */
29
34
  /** @typedef {import("../cli.mjs").LockHandle} LockHandle */
30
35
  /** @typedef {import("../contract/index.mjs").NodeSnapshot} NodeSnapshot */
31
36
  /** @typedef {import("../contract/index.mjs").RunMetadata} RunMetadata */
@@ -182,6 +187,25 @@ export function setLaunchBaseRef(baseRef) {
182
187
  pendingLaunchBaseRef = baseRef ?? null;
183
188
  }
184
189
 
190
+ /**
191
+ * Whether this launch must ask every routed runtime again even where the
192
+ * verdict store holds a fresh answer (`--fresh-preflight`). A module rather
193
+ * than an option for the same reason `setLaunchBaseRef` is: the gate is
194
+ * reached through `runContract` and `resumeRun`, which thread no launch
195
+ * options of their own. A forced launch still records what it observes.
196
+ *
197
+ * @type {boolean}
198
+ */
199
+ let pendingFreshPreflight = false;
200
+
201
+ /**
202
+ * @param {boolean} force
203
+ * @returns {void}
204
+ */
205
+ export function setFreshPreflight(force) {
206
+ pendingFreshPreflight = force === true;
207
+ }
208
+
185
209
  /**
186
210
  * The base ref a run was launched against, when it was launched with
187
211
  * `--base-ref`. A run recorded before this field existed, or launched
@@ -401,32 +425,207 @@ export function serializableContract(contract) {
401
425
  };
402
426
  }
403
427
  /**
404
- * The dispatch gate: no node starts until the host can carry the run. The
405
- * report is written as run evidence either way, and a blocking failure leaves
406
- * the materialized run untouched — the operator fixes the host and resumes,
407
- * so a run is never silently restarted and already-paid nodes are not redone.
428
+ * Live failure codes that name the pipeline rather than the provider, so they
429
+ * are verdicts of nothing and are never recorded: the two silences above, and
430
+ * `command_invalid`, which never reached a provider at all. The recordable
431
+ * set is the complement of this one, not of `LIVE_SILENCE_CAUSES` -- a
432
+ * command that could not be constructed passes the gate (validation owns
433
+ * that defect) but learned nothing about availability, so there is no verdict
434
+ * to persist.
435
+ */
436
+ const LIVE_NO_VERDICT_CAUSES = new Set(["preflight_timeout", "spawn_error", "command_invalid"]);
437
+
438
+ /**
439
+ * Whether an asked probe reached a provider and so carries a verdict worth
440
+ * recording. `done` reached it and completed; a failure whose detail names a
441
+ * live error code reached it too unless the code is one of the pipeline's own
442
+ * (see `LIVE_NO_VERDICT_CAUSES`). A refusal, an auth failure, unparsable
443
+ * output -- anything a provider itself produced -- is an answer.
444
+ *
445
+ * @param {ProbeResult} probe
446
+ * @returns {boolean}
447
+ */
448
+ function liveVerdictRecorded(probe) {
449
+ if (probe.liveStatus === "done") return true;
450
+ const match = / · live \S+ · ([a-z_]+):/u.exec(probe.detail ?? "");
451
+ return match !== null && !LIVE_NO_VERDICT_CAUSES.has(match[1]);
452
+ }
453
+
454
+ /**
455
+ * The dispatch gate: no node starts until the host can carry the run and the
456
+ * runtimes it routes to have answered. The static half proves the host facts
457
+ * — disk, git, worktree, a versioned binary per routed runtime. The live half
458
+ * asks every routed runtime one trivial prompt through `preflightContract`,
459
+ * read-only in a throwaway repository, because a present, versioned binary
460
+ * can still hold a dead credential, a spent quota, or a model that no longer
461
+ * answers — and each of those fails a run minutes in, after a worktree and a
462
+ * campaign event already exist.
463
+ *
464
+ * Answered means answered, not healthy: any verdict a provider returns,
465
+ * a quota refusal included, counts as having answered, and the run proceeds
466
+ * onto whatever the contract declares. Only pipeline silence blocks, and a
467
+ * silent runtime is named with cause unknown — the verdict
468
+ * `normalizeProviderAvailability` reserves for a probe that named no cause.
469
+ *
470
+ * The report is written as run evidence either way, and a blocking failure
471
+ * leaves the materialized run untouched — the operator fixes the host and
472
+ * resumes, so a run is never silently restarted and already-paid nodes are
473
+ * not redone.
474
+ *
475
+ * A verdict is no longer single-launch property: before asking, the gate
476
+ * reads the verdict store under the operator's home (`run/availability.mjs`),
477
+ * and a provider answered inside its window is reused, the evidence naming
478
+ * the reuse and the instant the verdict was observed. Every ask that reached
479
+ * a provider is recorded there for the next launch. `--fresh-preflight`
480
+ * skips the read; it never skips the record.
408
481
  *
409
482
  * @param {ValidatedContract} contract
410
483
  * @param {string} runDir
411
484
  * @param {SourceIdentity|undefined} sourceIdentity
412
485
  */
413
- export function assertEnvironmentReady(contract, runDir, sourceIdentity) {
486
+ export async function assertEnvironmentReady(contract, runDir, sourceIdentity) {
487
+ const at = new Date().toISOString();
414
488
  const report = environmentPreflight({
415
489
  cwd: contract.cwd,
416
490
  runtimes: reachableRuntimes(contract),
417
491
  harnessVersions: sourceIdentity?.harnessVersions ?? {},
418
492
  });
419
- const evidence = {
493
+ /** @param {boolean} ok @param {import("../harnesses/index.mjs").ProbeResult[]} [probes] @returns {Record<string, unknown>} */
494
+ const evidence = (ok, probes) => ({
420
495
  schemaVersion: report.schemaVersion,
421
496
  contractVersion: CONTRACT_VERSION,
422
- at: new Date().toISOString(),
497
+ at,
423
498
  contractId: contract.id,
424
- ok: report.ok,
499
+ ok,
425
500
  checks: report.checks,
426
- };
427
- writeJsonAtomic(join(runDir, "env-preflight.json"), evidence);
428
- if (report.ok) return;
429
- appendJsonl(join(runDir, "events.jsonl"), { type: "run.env-preflight-failed", ...evidence });
430
- const blocking = blockingChecks(report).map((check) => `${check.name}: ${check.detail}`).join(" · ");
501
+ ...(probes === undefined ? {} : { runtimes: probes }),
502
+ });
503
+ writeJsonAtomic(join(runDir, "env-preflight.json"), evidence(report.ok));
504
+ let blocking = report.ok ? null : blockingChecks(report).map((check) => `${check.name}: ${check.detail}`).join(" · ");
505
+ // A launch with nothing left to dispatch asks nothing: when every persisted
506
+ // state reads done, this launch replays an accepted transaction, starts no
507
+ // worker and no judge, and can spend no availability.
508
+ if (blocking === null && launchMayDispatch(runDir)) {
509
+ // measured 2026-09-22: asking four routed runtimes in parallel took about
510
+ // 18s, so the default budget is 60s. FABERUN_PREFLIGHT_TIMEOUT_SEC stays
511
+ // the operator override; preflightContract validates it, so only a valid
512
+ // number is lifted here.
513
+ const override = Number(process.env.FABERUN_PREFLIGHT_TIMEOUT_SEC);
514
+ const timeoutSec = process.env.FABERUN_PREFLIGHT_TIMEOUT_SEC !== undefined && Number.isFinite(override) && override > 0 ? override : 60;
515
+ const probes = await livePreflightProbes(contract, runDir, sourceIdentity, timeoutSec);
516
+ const silent = probes.filter((probe) => liveSilenceCause(probe) !== null);
517
+ writeJsonAtomic(join(runDir, "env-preflight.json"), evidence(silent.length === 0, probes));
518
+ if (silent.length > 0) {
519
+ blocking = `no runtime answered the live preflight: ${silent.map((probe) => `${probe.id ?? probe.harness} (cause unknown)`).join(" · ")}`;
520
+ }
521
+ }
522
+ if (blocking === null) return;
523
+ // The blocking failure is durable run evidence. The event carries exactly
524
+ // these seven fields — the live ProbeResults stay in env-preflight.json and
525
+ // never enter the event stream — and the append stays in this function: the
526
+ // field-ownership document names assertEnvironmentReady as the writer.
527
+ appendJsonl(join(runDir, "events.jsonl"), {
528
+ type: "run.env-preflight-failed",
529
+ schemaVersion: report.schemaVersion,
530
+ contractVersion: CONTRACT_VERSION,
531
+ at,
532
+ contractId: contract.id,
533
+ ok: false,
534
+ checks: report.checks,
535
+ });
431
536
  throw Object.assign(new Error(`env_preflight_failed: ${blocking} · the run stays resumable: fix the environment and resume ${runDir}`), { code: "env_preflight_failed" });
432
537
  }
538
+
539
+ /**
540
+ * The live half of the gate for one launch. Every routed runtime either holds
541
+ * a verdict this machine recorded inside its freshness window -- reused, the
542
+ * evidence naming the instant it was observed -- or is asked now, and every
543
+ * ask that reached a provider is recorded for the next launch. Nothing here
544
+ * decides what an answer means: reuse changes only whether the provider is
545
+ * asked, never whether a pass is a pass.
546
+ *
547
+ * The read keys on what identifies the provider -- harness, model, the
548
+ * executable the harness adapter itself resolves -- the same resolution a
549
+ * probe reports, so a runtime re-labelled between contracts is still one
550
+ * provider and a provider pointing at another binary is a new question.
551
+ *
552
+ * @param {ValidatedContract} contract
553
+ * @param {string} runDir
554
+ * @param {SourceIdentity|undefined} sourceIdentity
555
+ * @param {number} timeoutSec
556
+ * @returns {Promise<ProbeResult[]>}
557
+ */
558
+ async function livePreflightProbes(contract, runDir, sourceIdentity, timeoutSec) {
559
+ const routed = reachableRuntimes(contract);
560
+ /** @type {Map<string, RuntimeAvailability>} */
561
+ const fresh = new Map();
562
+ if (!pendingFreshPreflight) {
563
+ for (const [id, { runtime }] of routed) {
564
+ const verdict = readAvailability(availabilityKey({
565
+ harness: runtime.harness,
566
+ model: runtime.model,
567
+ executable: getHarness(runtime.harness).executable(runtime),
568
+ }));
569
+ if (verdict) fresh.set(id, verdict);
570
+ }
571
+ }
572
+ if (routed.size > 0 && fresh.size === routed.size) {
573
+ return [...routed.entries()].map(([id, { runtime }]) => {
574
+ const verdict = /** @type {RuntimeAvailability} */ (fresh.get(id));
575
+ return {
576
+ id,
577
+ harness: runtime.harness,
578
+ executable: getHarness(runtime.harness).executable(runtime),
579
+ model: runtime.model,
580
+ version: sourceIdentity?.harnessVersions?.[id] ?? null,
581
+ capabilities: harnessCapabilities(runtime),
582
+ requiredCapabilities: {},
583
+ requiredCapabilitySets: [],
584
+ ok: true,
585
+ live: true,
586
+ liveStatus: "reused",
587
+ detail: `live verdict reused · observed ${verdict.observedAt ?? "unknown instant"}`,
588
+ };
589
+ });
590
+ }
591
+ const probes = await preflightContract(join(runDir, "contract.json"), { liveTimeoutSec: timeoutSec, persisted: true });
592
+ // What this launch bought is durable from here on: every ask that reached a
593
+ // provider -- a refusal included, an answer being an answer -- is recorded
594
+ // under the provider's own identity. Silence and a command that never
595
+ // reached a provider are verdicts of nothing and are never recorded, so the
596
+ // operator who fixes the host is never told the fix "already answered".
597
+ recordAvailability(
598
+ probes
599
+ .filter((probe) => probe.live === true && probe.liveStatus !== "reused" && liveVerdictRecorded(probe))
600
+ .map((probe) => availabilityKey(probe)),
601
+ Date.now(),
602
+ );
603
+ return probes;
604
+ }
605
+
606
+ /**
607
+ * Whether this launch can dispatch anything. Read defensively: a missing
608
+ * nodes directory or an unparseable state means the launch may dispatch, so
609
+ * the runtimes are asked.
610
+ *
611
+ * @param {string} runDir
612
+ * @returns {boolean}
613
+ */
614
+ function launchMayDispatch(runDir) {
615
+ let names;
616
+ try {
617
+ names = readdirSync(join(runDir, "nodes"));
618
+ } catch {
619
+ return true;
620
+ }
621
+ const states = names.filter((name) => name.endsWith(".json"));
622
+ if (states.length === 0) return true;
623
+ return states.some((name) => {
624
+ try {
625
+ return JSON.parse(readFileSync(join(runDir, "nodes", name), "utf8")).status !== "done";
626
+ } catch {
627
+ return true;
628
+ }
629
+ });
630
+ }
631
+
@@ -60,7 +60,18 @@ export const DISCOVERY_RUNTIME_DEFINITIONS = Object.freeze({
60
60
  costRank: 1,
61
61
  },
62
62
  "agy-gemini": { harness: "agy", model: "gemini-3.8-flash-low", vendor: "google", tier: 1, costRank: 1 },
63
+ // Three codex rows, because the harness declares three models and an
64
+ // account is entitled to only some of them: measured 2026-09-21 on the
65
+ // owner's ChatGPT account, plain `gpt-5.6` answers HTTP 400 while
66
+ // `gpt-5.6-sol` answers normally. One row meant `setup` could offer only the
67
+ // model that account cannot use, and the operator's only way out was to hand
68
+ // every contract its own catalogue through `--runtimes`. Declaration order is
69
+ // unchanged, so the composed default is still `codex-gpt`: which of the three
70
+ // an account can reach is not something this file can know, and the live
71
+ // preflight is what reports it per id.
63
72
  "codex-gpt": { harness: "codex", model: "gpt-5.6", vendor: "openai", tier: 2, costRank: 2 },
73
+ "codex-sol": { harness: "codex", model: "gpt-5.6-sol", vendor: "openai", tier: 2, costRank: 2 },
74
+ "codex-luna": { harness: "codex", model: "gpt-5.6-luna", vendor: "openai", tier: 2, costRank: 2 },
64
75
  "claude-sonnet": { harness: "claude", model: "claude-sonnet-5", vendor: "anthropic", tier: 2, costRank: 2 },
65
76
  });
66
77
 
@@ -234,21 +245,42 @@ function tierOrder(runtime) {
234
245
  }
235
246
 
236
247
  /**
237
- * Span in seconds of every rate-limit window label a harness reports. The
238
- * labels are claude's `rateLimitType` values (measured 2026-09-17, the
239
- * `rate_limit_event` line recorded in `src/harnesses/protocol.mjs`); a label
248
+ * How long a recorded live-preflight verdict stays fresh, in seconds. This is
249
+ * the hello's own clock and is never derived from the quota windows below: a
250
+ * spend allowance expires when the provider resets it, a hello expires
251
+ * because whatever it proved -- a working credential, a spawning binary, a
252
+ * model that answers -- has stopped holding. Measured 2026-09-22: the ask
253
+ * costs about 18s for four runtimes in parallel, so every launch that reuses
254
+ * instead of asking saves about that. The window bets that a provider which
255
+ * answered still answers for the next quarter hour -- long enough to cover a
256
+ * burst of relaunches and retries, short enough that whatever died in
257
+ * between is bought again within fifteen minutes.
258
+ */
259
+ const PREFLIGHT_FRESH_SEC = 15 * 60;
260
+
261
+ /** The window label a persisted live-preflight verdict carries. */
262
+ export const PREFLIGHT_WINDOW = "preflight";
263
+
264
+ /**
265
+ * Span in seconds of every window label a catalogue or stored record may
266
+ * carry. `five_hour` and `seven_day` are claude's rate-limit labels (measured
267
+ * 2026-09-17, the `rate_limit_event` line recorded in
268
+ * `src/harnesses/protocol.mjs`); `preflight` is the hello's own clock, whose
269
+ * duration and reasoning live in `PREFLIGHT_FRESH_SEC` above -- the two kinds
270
+ * of window expire for different reasons and are never one clock. A label
240
271
  * missing here cannot prove staleness, so its observation never self-expires.
241
272
  *
242
273
  * @type {Readonly<Record<string, number>>}
243
274
  */
244
- const AVAILABILITY_WINDOW_SEC = Object.freeze({ five_hour: 5 * 3600, seven_day: 7 * 86400 });
275
+ const AVAILABILITY_WINDOW_SEC = Object.freeze({ five_hour: 5 * 3600, seven_day: 7 * 86400, [PREFLIGHT_WINDOW]: PREFLIGHT_FRESH_SEC });
245
276
 
246
277
  /**
247
278
  * May a runtime be admitted on this catalogue record? Exhaustion is waited
248
279
  * out on `exhaustedUntil`; an observation older than its own window reads as
249
280
  * unknown and admits nothing, because unknown must not look rested. This is
250
281
  * the one home of the rule: plan routing and engine composition both read it,
251
- * so the null and staleness semantics cannot drift between readers. The
282
+ * as does the verdict store's reuse decision in `run/availability.mjs`, so
283
+ * the null and staleness semantics cannot drift between readers. The
252
284
  * parameter is typed on the fields the rule reads, not on the full record --
253
285
  * the plan's table copy names no `reason`.
254
286
  *
@@ -37,13 +37,11 @@ import { delay, errorCode } from "../util.mjs";
37
37
  import { alreadyNotified, emitNodeAdvisories, notifyQueueFor, notifyQueuesByRun, renderCampaignHandoffSafely } from "./notify-queue.mjs";
38
38
  import { detectStalls, invocationAlive, terminateProcess } from "./process.mjs";
39
39
  import { transition, writeNode } from "./state.mjs";
40
- import { listNodeSnapshots, readNodeSnapshot } from "../run/node-store.mjs";
41
40
  import { render, renderFinalReport, writeFindingsArtifact } from "../report/final.mjs";
42
41
  import { operationNextState, providerReceipts, settleInvocation } from "../run/operations.mjs";
43
42
  import { appendUsageRecord, invocationCost, invocationUsage, recordInvocationUsage } from "../run/usage.mjs";
44
43
  import { captureNodeScopeBoundaries, checkWorkerScope, emptyScope } from "./scope.mjs";
45
44
  import { validateContractForLaunch } from "../campaign/chain.mjs";
46
- import { validateNodeSnapshot } from "../contract/snapshot.mjs";
47
45
  import { finalVerificationCommands, sharedVerificationCommands } from "../contract/final-verification.mjs";
48
46
  import { startJudge, startWorker } from "./dispatch.mjs";
49
47
  import { assertEnvironmentReady, captureRunIdentity, createRunMetadata, serializableContract, statesFingerprint } from "./run-identity.mjs";
@@ -238,6 +236,19 @@ export async function runContract(contractPath, options = {}) {
238
236
  }
239
237
  const lock = acquireLock(runDir);
240
238
  try {
239
+ // `contract.json` is written before anything else claims a name outside
240
+ // this directory, because `cancel` is the only verb that releases those
241
+ // names and it reads the contract from here. Everything below can fail or
242
+ // be killed -- `runtimeAssignments` probes providers, `captureRunIdentity`
243
+ // shells out to git, `createRunRef` claims `refs/faberun/<id>/run` -- and a
244
+ // launch that died between the ref and this write used to leave an
245
+ // occupied ref plus a run directory `cancel` could not parse: the operator
246
+ // was refused the directory, deleted it, was then refused the ref, and had
247
+ // no single verb for either. Written first, the directory is always
248
+ // cancellable from the instant it exists.
249
+ mkdirSync(join(runDir, "nodes"), { recursive: true });
250
+ mkdirSync(join(runDir, "logs"), { recursive: true });
251
+ writeJsonAtomic(join(runDir, "contract.json"), serializableContract(contract));
241
252
  const runtimePlan = await runtimeAssignments(contract);
242
253
  const scopeBoundaries = captureNodeScopeBoundaries(contract);
243
254
  const sourceIdentity = await captureRunIdentity(contract, scopeBoundaries);
@@ -245,9 +256,6 @@ export async function runContract(contractPath, options = {}) {
245
256
  lock.assert();
246
257
  const runsDir = runsRoot(contract.cwd);
247
258
  const campaign = resolveCampaign(runsDir, contract.campaignId);
248
- mkdirSync(join(runDir, "nodes"), { recursive: true });
249
- mkdirSync(join(runDir, "logs"), { recursive: true });
250
- writeJsonAtomic(join(runDir, "contract.json"), serializableContract(contract));
251
259
  writeJsonAtomic(join(runDir, "judge.schema.json"), JUDGE_SCHEMA);
252
260
  writeJsonAtomic(join(runDir, "run.json"), createRunMetadata(lock, sourceIdentity, {}, integrationRef));
253
261
  registerRun(campaign.path, contract.id);
@@ -315,7 +323,7 @@ export async function runContract(contractPath, options = {}) {
315
323
  */
316
324
  export async function driveRun(contract, runDir, states, campaign, lock, sourceIdentity, resume = {}, options = {}) {
317
325
  lock.assert();
318
- assertEnvironmentReady(contract, runDir, sourceIdentity);
326
+ await assertEnvironmentReady(contract, runDir, sourceIdentity);
319
327
  const runsDir = runsRoot(contract.cwd);
320
328
  const bootstrapNonce = bootstrapNonceForProcess();
321
329
  // Only the CLI entry can answer this: a nonce inherited by evals/run.mjs or
@@ -755,20 +763,3 @@ export async function driveRun(contract, runDir, states, campaign, lock, sourceI
755
763
  }
756
764
  return { runDir, states, ok: failed.length === 0 };
757
765
  }
758
-
759
- /**
760
- * @param {string} runDir
761
- * @param {ValidatedContract} contract
762
- * @returns {NodeSnapshot[]}
763
- */
764
- export function readRunNodes(runDir, contract) {
765
- const names = listNodeSnapshots(runDir);
766
- const expected = new Map(contract.nodes.map((node) => [`${node.id}.json`, node]));
767
- for (const name of names) if (!expected.has(name)) throw new TypeError(`unexpected persisted node snapshot ${name}`);
768
- return contract.nodes.map((node) => {
769
- const name = `${node.id}.json`;
770
- if (!names.includes(name)) throw new TypeError(`missing persisted node snapshot ${name}`);
771
- return validateNodeSnapshot(readNodeSnapshot(runDir, node.id), node);
772
- });
773
- }
774
-
@@ -15,7 +15,7 @@ import { SETTLED } from "./prompts.mjs";
15
15
  import { appendTransitionEvent, recordExecutionOverride, transition, writeNode } from "./state.mjs";
16
16
  import { attemptWorkspace } from "../repo/worktree.mjs";
17
17
 
18
- import { errorCode, errorMessage, excerpt, isContained } from "../util.mjs";
18
+ import { errorCode, errorMessage, excerpt, isContained, shellWords } from "../util.mjs";
19
19
  import { executeControllerVerification } from "./verify.mjs";
20
20
  import { providerReceiptsFromInvocationTail, settleInvocation } from "../run/operations.mjs";
21
21
  import { readJson } from "../run/store.mjs";
@@ -108,10 +108,17 @@ export function workerScope(taskPacket) {
108
108
 
109
109
  /**
110
110
  * Everything a node's own proofs name: a Definition of Done `path` proof's
111
- * path, the words of a `command` proof (a command proof carries a display
112
- * string, not an argv, so a quoted path holding a space is not recovered), the
113
- * argv of the verification entry a `verification` proof references, and the
114
- * argv of every verification command the packet declares.
111
+ * path, the words of a `command` proof, the argv of the verification entry a
112
+ * `verification` proof references, and the argv of every verification command
113
+ * the packet declares.
114
+ *
115
+ * A command proof carries a display string, not an argv, so its words are
116
+ * recovered with `shellWords` rather than a whitespace split -- the split lost
117
+ * exactly what the shell it runs under preserves. Measured 2026-09-22:
118
+ * `spawn(ref, {shell: true})` hands the whole string to `sh -c`, which groups
119
+ * `"b c"` into one argument, while the split here cut it into `"b` and `c"`,
120
+ * so a proof naming a real path with a space cited two fragments that matched
121
+ * no file and the write it excused read as unexpected.
115
122
  *
116
123
  * @param {ValidatedNode} node
117
124
  * @returns {ProofCitation[]}
@@ -129,7 +136,7 @@ function proofCitations(node) {
129
136
  if (proof.kind === "path") {
130
137
  citations.push({ tokens: [proof.ref], cwd: ".", literal: true, citation: `${item.id} path proof` });
131
138
  } else if (proof.kind === "command") {
132
- citations.push({ tokens: proof.ref.split(/\s+/u), cwd: ".", literal: false, citation: `${item.id} command proof` });
139
+ citations.push({ tokens: shellWords(proof.ref), cwd: ".", literal: false, citation: `${item.id} command proof` });
133
140
  } else {
134
141
  const command = commands[Number.parseInt(proof.ref, 10)];
135
142
  if (command) citations.push({ tokens: command.argv, cwd: command.cwd ?? ".", literal: false, citation: `${item.id} verification[${proof.ref}] proof` });