faberun 0.19.1 → 0.19.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/cli/plan.mjs CHANGED
@@ -4,8 +4,8 @@
4
4
  * contested. This file only owns the wire — `src/plan/pipeline.mjs` owns the
5
5
  * sequencing and every decision the pipeline makes.
6
6
  */
7
- import { readFileSync } from "node:fs";
8
- import { resolve } from "node:path";
7
+ import { mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
8
+ import { dirname, join, resolve } from "node:path";
9
9
  import { detachArgv, detachSelf, waitForBootstrap } from "./launch.mjs";
10
10
  import { classifyRunProgress } from "../campaign/chain.mjs";
11
11
  import { runProgress } from "../engine/supervise.mjs";
@@ -15,11 +15,23 @@ import { validateFinalVerification, validateSharedVerification } from "../contra
15
15
  import { colorLevel, statusToken } from "./brand.mjs";
16
16
  import { delay } from "../util.mjs";
17
17
  import { runPlanningPipeline } from "../plan/pipeline.mjs";
18
- import { runDirectory } from "../run/paths.mjs";
18
+ import { campaignTree, runDirectory } from "../run/paths.mjs";
19
+ import { readCampaign } from "../campaign/record.mjs";
19
20
 
20
21
  /** How often a foreground `plan` polls a launched stage's run directory. */
21
22
  const DEFAULT_POLL_MS = 1_000;
22
23
 
24
+ /**
25
+ * How long a `--detach` launcher stays to see whether the planning process it
26
+ * started is actually up. Long enough to cover the bootstrap work that fails
27
+ * synchronously -- spec read, strict validation, catalogue load, the runtime
28
+ * ask -- and short enough that a launcher is not a supervisor: past this
29
+ * window, a planning run that dies is a running campaign's problem and leaves
30
+ * its evidence in the run directory, not here.
31
+ */
32
+ const PLAN_BOOTSTRAP_WINDOW_MS = 5_000;
33
+ const PLAN_BOOTSTRAP_POLL_MS = 100;
34
+
23
35
  /**
24
36
  * `--runtime-defaults worker=<id>,judge=<id>`, either key optional, comma
25
37
  * separated. Absent entirely, the pipeline falls through to plain
@@ -114,7 +126,7 @@ export function loadVerificationSuites(path) {
114
126
 
115
127
  /**
116
128
  * @param {string} target
117
- * @param {{campaign?: string, phase?: string, "review-rounds"?: string, "approve-below"?: string, "runtime-defaults"?: string, runtimes?: string, verification?: string, detach?: boolean, json?: boolean}} values
129
+ * @param {{campaign?: string, phase?: string, "review-rounds"?: string, "approve-below"?: string, "runtime-defaults"?: string, runtimes?: string, verification?: string, package?: string, "targeted-fix"?: boolean, detach?: boolean, json?: boolean}} values
118
130
  * @returns {Promise<void>}
119
131
  */
120
132
  export async function planCli(target, values) {
@@ -131,6 +143,7 @@ export async function planCli(target, values) {
131
143
  const verification = typeof values.verification === "string" && values.verification
132
144
  ? loadVerificationSuites(values.verification)
133
145
  : {};
146
+ const packageMode = packageModeOf(values.package);
134
147
 
135
148
  if (values.detach === true) {
136
149
  const argv = ["plan", specPath, "--campaign", campaignId, "--phase", phase, "--review-rounds", String(reviewRounds)];
@@ -138,35 +151,71 @@ export async function planCli(target, values) {
138
151
  if (values["runtime-defaults"] !== undefined) argv.push("--runtime-defaults", values["runtime-defaults"]);
139
152
  if (typeof values.runtimes === "string" && values.runtimes) argv.push("--runtimes", resolve(values.runtimes));
140
153
  if (typeof values.verification === "string" && values.verification) argv.push("--verification", resolve(values.verification));
154
+ if (packageMode !== "implementation") argv.push("--package", packageMode);
155
+ if (values["targeted-fix"] === true) argv.push("--targeted-fix");
156
+ const failurePath = planBootstrapFailurePath(process.cwd(), campaignId, phase);
157
+ // Read the campaign before creating anything: the failure record lives
158
+ // inside the campaign tree, so a typo in --campaign would otherwise leave
159
+ // a campaign directory with no record in it for `discoverCampaigns` to
160
+ // find. The child reads it too; this is the launcher refusing what it can
161
+ // see for itself rather than detaching into a certain failure.
162
+ readCampaign(campaignTree(process.cwd(), campaignId));
163
+ mkdirSync(dirname(failurePath), { recursive: true });
164
+ rmSync(failurePath, { force: true });
141
165
  const child = detachArgv(argv);
142
166
  if (child.pid === undefined) throw new Error("detached plan has no pid");
167
+ // The child's stdio is discarded (detachArgv), so a planning run that dies
168
+ // during bootstrap used to take its own reason with it while the launcher
169
+ // had already printed a pid and exited 0. Measured 2026-09-21 on macOS and
170
+ // Linux the same day: a run died after collecting repo facts and the stderr
171
+ // went with the closed connection. Waiting out a bounded window is the
172
+ // whole check -- a controller still alive past it is up, and one that is
173
+ // not has written why.
174
+ const failure = await watchPlanBootstrap(child, failurePath);
175
+ if (failure) {
176
+ process.stderr.write(`[plan] bootstrap failed · ${failure.error}\n[plan] recorded at ${failurePath}\n`);
177
+ process.exitCode = 1;
178
+ return;
179
+ }
143
180
  process.stdout.write(`[plan] detached · pid ${child.pid} · ${specPath}\n`);
144
181
  return;
145
182
  }
146
183
 
147
- const result = await runPlanningPipeline({
148
- specPath,
149
- campaignId,
150
- phase,
151
- reviewRounds,
152
- approveBelow,
153
- runtimeDefaults,
154
- runtimes,
155
- verification,
156
- launch: async (contractPath, contract) => {
157
- const child = detachSelf("run", contractPath);
158
- if (child.pid === undefined) throw new Error("detached planning run has no pid");
159
- await waitForBootstrap(runDirectory(contract.cwd, contract.id), child.pid, child);
160
- },
161
- wait: async (runDir) => {
162
- for (;;) {
163
- const progress = runProgress(runDir);
164
- const classification = classifyRunProgress(progress);
165
- if (classification !== "unfinished" && classification !== "waiting") return progress;
166
- await delay(DEFAULT_POLL_MS);
167
- }
168
- },
169
- });
184
+ // A detached child arrives here with its stdio already discarded, so what
185
+ // it throws reaches nobody unless it is written down first. The record is
186
+ // written on every path, detached or not: a foreground failure that also
187
+ // leaves the file costs nothing and reads the same.
188
+ let result;
189
+ try {
190
+ result = await runPlanningPipeline({
191
+ specPath,
192
+ campaignId,
193
+ phase,
194
+ reviewRounds,
195
+ approveBelow,
196
+ runtimeDefaults,
197
+ runtimes,
198
+ verification,
199
+ targetedFix: values["targeted-fix"] === true,
200
+ packageMode,
201
+ launch: async (contractPath, contract) => {
202
+ const child = detachSelf("run", contractPath);
203
+ if (child.pid === undefined) throw new Error("detached planning run has no pid");
204
+ await waitForBootstrap(runDirectory(contract.cwd, contract.id), child.pid, child);
205
+ },
206
+ wait: async (runDir) => {
207
+ for (;;) {
208
+ const progress = runProgress(runDir);
209
+ const classification = classifyRunProgress(progress);
210
+ if (classification !== "unfinished" && classification !== "waiting") return progress;
211
+ await delay(DEFAULT_POLL_MS);
212
+ }
213
+ },
214
+ });
215
+ } catch (error) {
216
+ writePlanBootstrapFailure(process.cwd(), campaignId, phase, error instanceof Error ? error : new Error(String(error)));
217
+ throw error;
218
+ }
170
219
 
171
220
  if (values.json === true) {
172
221
  process.stdout.write(`${JSON.stringify(result)}\n`);
@@ -180,3 +229,100 @@ export async function planCli(target, values) {
180
229
  for (const warning of result.warnings) process.stdout.write(`${statusToken("warn", colorLevel(process.env, process.stdout.isTTY))} ${warning}\n`);
181
230
  process.stdout.write(`[plan] ${campaignId} phase ${phase} frozen · approved ${result.approved} · ${result.contractPath}\n`);
182
231
  }
232
+
233
+ /**
234
+ * Where a detached planning run records why it never came up. It sits beside
235
+ * the phase's durable plan artifacts rather than in the disposable scratch
236
+ * tree, and it is derived from campaign and phase alone -- the launcher and
237
+ * the child compute the same path without either having to parse the spec.
238
+ *
239
+ * @param {string} cwd
240
+ * @param {string} campaignId
241
+ * @param {string} phase
242
+ * @returns {string}
243
+ */
244
+ export function planBootstrapFailurePath(cwd, campaignId, phase) {
245
+ return join(campaignTree(cwd, campaignId), "plans", phase, "bootstrap-failure.json");
246
+ }
247
+
248
+ /**
249
+ * Record a detached planning run's bootstrap failure where its launcher can
250
+ * read it. Best effort: a failure to write this must never replace the
251
+ * failure it was describing.
252
+ *
253
+ * @param {string} cwd
254
+ * @param {string} campaignId
255
+ * @param {string} phase
256
+ * @param {Error} error
257
+ * @returns {void}
258
+ */
259
+ export function writePlanBootstrapFailure(cwd, campaignId, phase, error) {
260
+ try {
261
+ const path = planBootstrapFailurePath(cwd, campaignId, phase);
262
+ mkdirSync(dirname(path), { recursive: true });
263
+ writeFileSync(path, `${JSON.stringify({ at: new Date().toISOString(), pid: process.pid, campaignId, phase, error: error.message }, null, 2)}\n`);
264
+ } catch {
265
+ // Nothing left to do: the caller is already reporting the real failure.
266
+ }
267
+ }
268
+
269
+ /**
270
+ * Watch a freshly detached planning child through its bootstrap window.
271
+ *
272
+ * Returns the recorded failure when the child died inside the window, and
273
+ * null when it is still running at the end of it. A child that exits zero
274
+ * inside the window also reads as no failure: a planning run can legitimately
275
+ * be that fast only by refusing early, and it will have written its own
276
+ * record if it refused.
277
+ *
278
+ * @param {import("node:child_process").ChildProcess} child
279
+ * @param {string} failurePath
280
+ * @param {{windowMs?: number, pollMs?: number}} [options]
281
+ * @returns {Promise<{error: string}|null>}
282
+ */
283
+ export async function watchPlanBootstrap(child, failurePath, { windowMs = PLAN_BOOTSTRAP_WINDOW_MS, pollMs = PLAN_BOOTSTRAP_POLL_MS } = {}) {
284
+ let exitCode = /** @type {number|null|undefined} */ (undefined);
285
+ let exited = false;
286
+ child.once("exit", (code) => { exited = true; exitCode = code; });
287
+ const deadline = Date.now() + windowMs;
288
+ while (Date.now() < deadline) {
289
+ // Typed checks, not `!== null`: a child object that never carried the
290
+ // property at all would otherwise read as one that has already exited.
291
+ if (exited || typeof child.exitCode === "number" || typeof child.signalCode === "string") {
292
+ const recorded = readPlanBootstrapFailure(failurePath);
293
+ if (recorded) return recorded;
294
+ const code = typeof exitCode === "number" ? exitCode : child.exitCode;
295
+ if (typeof code === "number" && code !== 0) return { error: `the detached planning process exited ${code} without recording a reason` };
296
+ const signal = child.signalCode;
297
+ if (typeof signal === "string") return { error: `the detached planning process was killed by ${signal} without recording a reason` };
298
+ return null;
299
+ }
300
+ await delay(pollMs);
301
+ }
302
+ return null;
303
+ }
304
+
305
+ /** @param {string} failurePath @returns {{error: string}|null} */
306
+ function readPlanBootstrapFailure(failurePath) {
307
+ try {
308
+ const record = JSON.parse(readFileSync(failurePath, "utf8"));
309
+ return typeof record?.error === "string" ? { error: record.error } : null;
310
+ } catch {
311
+ return null;
312
+ }
313
+ }
314
+
315
+ /**
316
+ * `--package implementation|exploratory`. Implementation is the default and
317
+ * the only mode there was: nodes sized by their write set. Exploratory sizes
318
+ * by what a node reads, accepts a one-file write set as the normal shape of a
319
+ * finding, and reports a node whose read surface dwarfs its siblings'.
320
+ *
321
+ * @param {unknown} value
322
+ * @returns {import("../plan/sizing.mjs").PackageMode}
323
+ */
324
+ export function packageModeOf(value) {
325
+ if (value === undefined) return "implementation";
326
+ if (value === "implementation" || value === "exploratory") return value;
327
+ throw new Error(`--package must be implementation or exploratory: ${String(value)}`);
328
+ }
package/src/cli/spec.mjs CHANGED
@@ -14,7 +14,7 @@ import { validateSpec } from "../plan/spec.mjs";
14
14
  /** Flags are scoped to the operation that declares them; all others are rejected. */
15
15
  /** @type {Record<string, import("node:util").ParseArgsOptionsConfig>} */
16
16
  const OPERATION_OPTIONS = {
17
- validate: { "strict-traceability": { type: "boolean" }, json: { type: "boolean" } },
17
+ validate: { "strict-traceability": { type: "boolean" }, "run-proofs": { type: "boolean" }, json: { type: "boolean" } },
18
18
  scaffold: { id: { type: "string" } },
19
19
  };
20
20
 
@@ -63,9 +63,13 @@ export function specCli(args) {
63
63
  }
64
64
  const target = parsed.positionals[0];
65
65
  if (!target || parsed.positionals.length > 1) return usage();
66
- const values = /** @type {{"strict-traceability"?: boolean, json?: boolean, id?: string}} */ (parsed.values);
66
+ const values = /** @type {{"strict-traceability"?: boolean, "run-proofs"?: boolean, json?: boolean, id?: string}} */ (parsed.values);
67
67
  if (operation === "validate") {
68
- validateSpecFile(resolve(target), { strict: values["strict-traceability"] === true, json: values.json === true });
68
+ validateSpecFile(resolve(target), {
69
+ strict: values["strict-traceability"] === true,
70
+ runProofs: values["run-proofs"] === true,
71
+ json: values.json === true,
72
+ });
69
73
  return;
70
74
  }
71
75
  try {
@@ -80,12 +84,18 @@ export function specCli(args) {
80
84
  * Validate a spec file and print its class, its overall verdict, and one
81
85
  * line per finding. Exits `1` when the verdict is not `ok`.
82
86
  *
87
+ * `runProofs` is the only operation here that spawns anything: it runs each
88
+ * requirement's declared proof instead of only checking that one is written
89
+ * down. It is opt-in because running a repository's proofs costs real time,
90
+ * and default-off keeps `spec validate` the deterministic, side-effect-free
91
+ * read it has always been.
92
+ *
83
93
  * @param {string} path
84
- * @param {{strict: boolean, json: boolean}} options
94
+ * @param {{strict: boolean, json: boolean, runProofs?: boolean}} options
85
95
  * @returns {SpecValidation}
86
96
  */
87
- export function validateSpecFile(path, { strict, json }) {
88
- const result = validateSpec(readFileSync(path, "utf8"), { cwd: process.cwd(), strict });
97
+ export function validateSpecFile(path, { strict, json, runProofs = false }) {
98
+ const result = validateSpec(readFileSync(path, "utf8"), { cwd: process.cwd(), strict, runProofs });
89
99
  if (json) {
90
100
  process.stdout.write(`${JSON.stringify(result)}\n`);
91
101
  } else {
@@ -112,7 +122,7 @@ export function scaffoldSpec(path, id) {
112
122
 
113
123
  /** @returns {void} */
114
124
  function usage() {
115
- process.stderr.write("usage: faberun spec <validate|scaffold> <path> [--strict-traceability] [--json] [--id <value>]\n");
125
+ process.stderr.write("usage: faberun spec <validate|scaffold> <path> [--strict-traceability] [--run-proofs] [--json] [--id <value>]\n");
116
126
  process.exitCode = 2;
117
127
  }
118
128
 
package/src/cli.mjs CHANGED
@@ -123,6 +123,8 @@ export const COMMAND_OPTIONS = {
123
123
  "runtime-defaults": { type: "string" },
124
124
  runtimes: { type: "string" },
125
125
  verification: { type: "string" },
126
+ package: { type: "string" },
127
+ "targeted-fix": { type: "boolean" },
126
128
  detach: { type: "boolean" },
127
129
  json: { type: "boolean" },
128
130
  },
@@ -95,3 +95,53 @@ function validateVerificationRef(value, label, options) {
95
95
  return String(index);
96
96
  }
97
97
 
98
+ /**
99
+ * A `kind: "command"` proof's `ref` runs through a shell (`judge-gate.mjs`'s
100
+ * `proveCommand`), while a task packet's `verification` entries run as argv
101
+ * arrays -- the opposite quoting convention for the same author intent.
102
+ * Measured 2026-09-22: six DoD refs on one campaign wrote
103
+ * `--test-name-pattern=a b c` the way an argv array would take it, the shell
104
+ * split it into three words, and the gate rejected work whose identical
105
+ * command had just passed as verification. Nothing warned the author, because
106
+ * a shell that receives extra bare words after an unquoted flag value does not
107
+ * itself know they were meant to be one argument.
108
+ *
109
+ * This flags the same shape rather than every proof: a node:test filter flag
110
+ * (the flags `declaredTestFilters` in judge-gate.mjs recognizes) whose
111
+ * unquoted value is immediately followed by bare words is the pattern the
112
+ * incident measured, and it is precise enough that a legitimate `ref` rarely
113
+ * has trailing bare words right after such a flag by accident.
114
+ *
115
+ * @param {DefinitionOfDoneItem[]} items
116
+ * @param {number} index
117
+ * @returns {string[]}
118
+ */
119
+ export function unquotedFilterValueWarnings(items, index) {
120
+ /** @type {string[]} */
121
+ const warnings = [];
122
+ items.forEach((item, itemIndex) => {
123
+ if (item.proof?.kind !== "command") return;
124
+ const tokens = item.proof.ref.split(/\s+/u).filter(Boolean);
125
+ for (const flag of TEST_FILTER_FLAGS) {
126
+ for (let position = 0; position < tokens.length; position++) {
127
+ const prefix = `${flag}=`;
128
+ if (!tokens[position].startsWith(prefix)) continue;
129
+ const value = tokens[position].slice(prefix.length);
130
+ if (/^['"]/u.test(value)) continue;
131
+ let end = position + 1;
132
+ while (end < tokens.length && !tokens[end].startsWith("-")) end++;
133
+ if (end === position + 1) continue;
134
+ const spilled = [value, ...tokens.slice(position + 1, end)].join(" ");
135
+ warnings.push(
136
+ `nodes[${index}] (definitionOfDone[${itemIndex}]): proof.ref's ${flag} value "${spilled}" is unquoted; ` +
137
+ `kind: "command" runs through a shell, unlike taskPacket.verification's argv, so the space splits it into extra words -- quote it as ${flag}="${spilled}"`,
138
+ );
139
+ }
140
+ }
141
+ });
142
+ return warnings;
143
+ }
144
+
145
+ /** The node:test filter flags a `kind: "command"` proof's shell can split on an unquoted value. */
146
+ const TEST_FILTER_FLAGS = ["--test-name-pattern", "--test-skip-pattern"];
147
+
@@ -60,6 +60,27 @@ export function sharedVerificationCommands(contract) {
60
60
  return contract.sharedVerification ?? [];
61
61
  }
62
62
 
63
+ /**
64
+ * The timeout a Definition of Done `command` proof is spawned with. It used
65
+ * to be capped at a hardcoded 120s regardless of what the node's own attempt
66
+ * was given, so a contract that raised `timeoutSec` to run a slow command as
67
+ * `taskPacket.verification` still had the identical command fail its DoD
68
+ * proof on the same run: measured 2026-09-22, six proofs rejected work whose
69
+ * argv had just passed with a `timeoutSec` well past 120s. A command proof
70
+ * gets the same budget the node's own attempt has, because that is the
71
+ * budget the author already reasoned about; nothing here invents a second
72
+ * number for the gate to disagree with. Shared by `dispatch.mjs` (the actual
73
+ * spawn) and `scheduler.mjs` (the freeze-detection budget that must not judge
74
+ * a node frozen before its own gate's timeout has had a chance to fire).
75
+ *
76
+ * @param {{timeoutSec?: number}} node
77
+ * @param {{timeoutSec?: number}} contract
78
+ * @returns {number}
79
+ */
80
+ export function gateProofTimeoutMs(node, contract) {
81
+ return Math.max(1_000, (node.timeoutSec ?? contract.timeoutSec ?? 60) * 1_000);
82
+ }
83
+
63
84
  /**
64
85
  * Whether no other node in the contract depends on this one. `finalVerification`
65
86
  * is a candidate to run on a phase-terminal node only; a node with a dependant
@@ -3,9 +3,9 @@ import { readFileSync, statSync } from "node:fs";
3
3
  import { dirname, resolve } from "node:path";
4
4
  import { loadTaskPacket, renderWorkerPrompt } from "./task-packet.mjs";
5
5
  import { RESERVED_ARTICLES } from "./articles.mjs";
6
- import { validateDefinitionOfDone } from "./definition-of-done.mjs";
6
+ import { unquotedFilterValueWarnings, validateDefinitionOfDone } from "./definition-of-done.mjs";
7
7
  import { validateFinalVerification, validateSharedVerification } from "./final-verification.mjs";
8
- import { VERIFICATION_LIMITS } from "./verification.mjs";
8
+ import { VERIFICATION_LIMITS, requirementProofWarnings } from "./verification.mjs";
9
9
  import {
10
10
  validateCapabilityRequirements,
11
11
  } from "../harnesses/index.mjs";
@@ -374,12 +374,24 @@ export function validateContract(raw, contractPath, options = {}) {
374
374
  const sharedVerification = validateSharedVerification(raw.sharedVerification, "contract.sharedVerification");
375
375
  const contractCommands = [...finalVerification ?? [], ...sharedVerification ?? []].map((command) => command.argv.join(" "));
376
376
  const contractWrites = new Set(nodes.flatMap((node) => node.taskPacket.writeFiles ?? []));
377
- const warnings = nodes.flatMap((node, index) => [
378
- ...commandCoverageWarnings(node, index),
379
- ...(persisted ? [] : mirrorCoverageWarnings(node, index, cwd, contractCommands, contractWrites)),
380
- ...(persisted ? [] : unsnapshottedWriteWarnings(node, index, cwd)),
381
- ...(persisted ? [] : writeFileLineBudgetWarnings(node, index, cwd)),
382
- ]);
377
+ const warnings = [
378
+ ...nodes.flatMap((node, index) => [
379
+ ...commandCoverageWarnings(node, index),
380
+ ...unquotedFilterValueWarnings(node.definitionOfDone ?? [], index),
381
+ ...(persisted ? [] : mirrorCoverageWarnings(node, index, cwd, contractCommands, contractWrites)),
382
+ ...(persisted ? [] : unsnapshottedWriteWarnings(node, index, cwd)),
383
+ ...(persisted ? [] : writeFileLineBudgetWarnings(node, index, cwd)),
384
+ ]),
385
+ // Cross-node by construction: a requirement proven in two nodes is only
386
+ // visible when every node's commands are read together, which is the
387
+ // whole point -- one copy repaired and six left behind is what a per-node
388
+ // read cannot see.
389
+ ...requirementProofWarnings([
390
+ ...nodes.map((node, index) => ({ id: `nodes[${index}] (${node.id})`, requirementIds: node.requirementIds, commands: node.taskPacket.verification ?? [] })),
391
+ { id: "contract.sharedVerification", commands: sharedVerification ?? [] },
392
+ { id: "contract.finalVerification", commands: finalVerification ?? [] },
393
+ ]),
394
+ ];
383
395
  const contract = /** @type {ValidatedContract} */ ({
384
396
  ...raw,
385
397
  schemaVersion: /** @type {number} */ (raw.schemaVersion),
@@ -8,7 +8,18 @@ export const VERIFICATION_LIMITS = Object.freeze({
8
8
  stderrBytes: 16 * 1024,
9
9
  maxCommands: 32,
10
10
  maxRepeat: 8,
11
- maxTimeoutSec: 600,
11
+ // A verification command may declare up to half an hour. It was 600s, which
12
+ // is below this repository's own suite: measured 2026-09-22, `npm test`
13
+ // takes 434-473s at the default parallelism on the author's machine and 650s
14
+ // on a Windows CI runner, and `node --test --test-concurrency=1 test/engine/`
15
+ // -- the way a packet's verification actually runs it -- takes 1035-1058s.
16
+ // So the one command that proves the engine could not be declared at all,
17
+ // and a packet author's way out was `--test-name-pattern`, which exits 0
18
+ // when it matches nothing (see AGENTS.md). A cap that pushes authors toward
19
+ // a proof that certifies nothing is worse than a longer runaway. The node's
20
+ // own wall clock (`contract.timeoutSec`, 2400s by default) still bounds the
21
+ // attempt above this.
22
+ maxTimeoutSec: 1_800,
12
23
  stateStdoutBytes: 2 * 1024,
13
24
  stateCommands: 16,
14
25
  stateAttempts: 4,
@@ -42,7 +53,14 @@ export const MUTATION_TIERS = Object.freeze({
42
53
  /**
43
54
  * One declared deterministic check: an argv command run by the controller.
44
55
  *
45
- * @typedef {{argv: string[], cwd?: string, timeoutSec?: number, repeat?: number, env?: string[], mutation?: {tier: MutationTier}}} VerificationCommand
56
+ * `requirementId` names the spec requirement this command is the proof of. It
57
+ * changes nothing about how the command runs; it is what makes a duplicated
58
+ * proof visible. Measured 2026-09-22: one broken command lived in a spec's R3,
59
+ * in its R4 and in seven nodes' verification, and the repair reached one of
60
+ * them — nothing could tell that the other copies had stopped agreeing,
61
+ * because nothing recorded that they were copies of one claim.
62
+ *
63
+ * @typedef {{argv: string[], cwd?: string, timeoutSec?: number, repeat?: number, env?: string[], mutation?: {tier: MutationTier}, requirementId?: string}} VerificationCommand
46
64
  */
47
65
 
48
66
  /**
@@ -122,7 +140,7 @@ export function validateVerificationCommands(commands, label = "verification") {
122
140
  function validateVerificationCommand(command, label = "verification command") {
123
141
  if (!command || typeof command !== "object" || Array.isArray(command)) throw new TypeError(`${label} must be an argv command object`);
124
142
  const record = /** @type {Record<string, unknown>} */ (command);
125
- const allowed = new Set(["argv", "cwd", "timeoutSec", "repeat", "env", "mutation"]);
143
+ const allowed = new Set(["argv", "cwd", "timeoutSec", "repeat", "env", "mutation", "requirementId"]);
126
144
  for (const key of Object.keys(record)) if (!allowed.has(key)) throw new TypeError(`${label} has unexpected field ${key}`);
127
145
  if (!Array.isArray(record.argv) || record.argv.length === 0 || record.argv.length > 64 || record.argv.some((item) => typeof item !== "string" || !item.trim() || Buffer.byteLength(item, "utf8") > 8 * 1024)) {
128
146
  throw new TypeError(`${label}.argv must be a non-empty array of strings`);
@@ -157,10 +175,18 @@ function validateVerificationCommand(command, label = "verification command") {
157
175
  }
158
176
  mutation = { tier: /** @type {MutationTier} */ (mutationRecord.tier) };
159
177
  }
178
+ // Bounded exactly like the node-level `requirementIds` it must match against
179
+ // (at most 128 bytes), so the two sides of the claim cannot accept different
180
+ // ids.
181
+ if (record.requirementId !== undefined
182
+ && (typeof record.requirementId !== "string" || !record.requirementId.trim() || Buffer.byteLength(record.requirementId, "utf8") > 128)) {
183
+ throw new TypeError(`${label}.requirementId must be a requirement id of at most 128 bytes`);
184
+ }
160
185
  /** @type {VerificationCommand} */
161
186
  const normalized = { argv: [.../** @type {string[]} */ (record.argv)], timeoutSec, repeat, env: [.../** @type {string[]} */ (env)] };
162
187
  if (record.cwd !== undefined) normalized.cwd = /** @type {string} */ (record.cwd);
163
188
  if (mutation !== undefined) normalized.mutation = mutation;
189
+ if (record.requirementId !== undefined) normalized.requirementId = /** @type {string} */ (record.requirementId);
164
190
  return normalized;
165
191
  }
166
192
 
@@ -199,3 +225,49 @@ export function compactVerification(result) {
199
225
  return { passed: Boolean(result?.passed), commands };
200
226
  }
201
227
 
228
+
229
+ /**
230
+ * One proof, one source. A command that declares `requirementId` says it is
231
+ * the proof of that requirement; this reports the two ways such a claim can
232
+ * be false.
233
+ *
234
+ * A claim the node does not carry is a mislabel: the node's own
235
+ * `requirementIds` are what the phase assigned it, and a command proving
236
+ * something outside them is either the wrong id or the wrong node.
237
+ *
238
+ * Copies that stopped agreeing are the measured one. 2026-09-22: a broken
239
+ * command lived in a spec's R3, its R4, and seven nodes' verification, and
240
+ * the repair reached one copy. Nothing could see that the others had drifted,
241
+ * because nothing recorded that they were copies of a single claim. Argv is
242
+ * compared against argv, never a joined string against a shell command: a
243
+ * joined argv loses argument boundaries, which is the same reason a
244
+ * `verification` proof references an index instead of comparing text.
245
+ *
246
+ * @param {Array<{id: string, requirementIds?: string[], commands: VerificationCommand[]}>} owners
247
+ * @returns {string[]}
248
+ */
249
+ export function requirementProofWarnings(owners) {
250
+ /** @type {string[]} */
251
+ const warnings = [];
252
+ /** @type {Map<string, Array<{owner: string, position: number, argv: string[]}>>} */
253
+ const claims = new Map();
254
+ for (const owner of owners) {
255
+ owner.commands.forEach((command, position) => {
256
+ const requirementId = command.requirementId;
257
+ if (requirementId === undefined) return;
258
+ if (owner.requirementIds !== undefined && !owner.requirementIds.includes(requirementId)) {
259
+ warnings.push(`${owner.id}: verification[${position}] declares requirementId "${requirementId}", which this node does not carry in requirementIds`);
260
+ }
261
+ const claimed = claims.get(requirementId) ?? [];
262
+ claimed.push({ owner: owner.id, position, argv: command.argv });
263
+ claims.set(requirementId, claimed);
264
+ });
265
+ }
266
+ for (const [requirementId, claimed] of claims) {
267
+ const distinct = new Map(claimed.map((claim) => [JSON.stringify(claim.argv), claim]));
268
+ if (distinct.size < 2) continue;
269
+ const listed = [...distinct.values()].map((claim) => `${claim.owner}: verification[${claim.position}] runs ${JSON.stringify(claim.argv)}`).join("; ");
270
+ warnings.push(`requirementId "${requirementId}" is proven by commands that no longer agree: ${listed}`);
271
+ }
272
+ return warnings;
273
+ }
@@ -25,6 +25,7 @@ import { attemptWorkspace, createAttemptWorktree, sealAttempt } from "../repo/wo
25
25
  import { attemptWorktreePath } from "../run/paths.mjs";
26
26
  import { basename, dirname, join } from "node:path";
27
27
  import { errorCode, errorMessage } from "../util.mjs";
28
+ import { gateProofTimeoutMs } from "../contract/final-verification.mjs";
28
29
  import { captureWorkspaceScope, captureWorkspaceSnapshot } from "../repo/workspace.mjs";
29
30
  import { deterministicGate, judgeReaskReason, judgeRequired, judgeSkippedByScope } from "./judge-gate.mjs";
30
31
  import { emptyScope, persistedScopeBoundary, workerScope } from "./scope.mjs";
@@ -459,7 +460,7 @@ export async function startJudge(contract, node, state, runDir, running, workerR
459
460
  node,
460
461
  workspace,
461
462
  reask,
462
- Math.max(1_000, Math.min((node.timeoutSec ?? contract.timeoutSec ?? 60) * 1000, 120_000)),
463
+ gateProofTimeoutMs(node, contract),
463
464
  /** @type {import("../contract/index.mjs").VerificationState|null} */ (state.verification),
464
465
  );
465
466
  state.review = reviewMode(node.gate);
@@ -25,7 +25,8 @@
25
25
  * It also exits when the release file's directory is gone (measured
26
26
  * 2026-09-16: gate processes from a prior day's test runs, spawned into a
27
27
  * temp directory the failed test never cleaned up, were still alive and
28
- * waiting for a release file that could now never appear).
28
+ * waiting for a release file that could now never appear), and it keeps
29
+ * asking both of those questions after the provider starts, not only before.
29
30
  */
30
31
  import { existsSync, readFileSync, statSync, openSync, closeSync, readSync, writeSync } from "node:fs";
31
32
  import { dirname } from "node:path";
@@ -186,11 +187,32 @@ function releaseDirectoryGone() {
186
187
  return !existsSync(dirname(releasePath));
187
188
  }
188
189
 
190
+ /**
191
+ * How often the gate re-asks its two liveness questions once the provider is
192
+ * running. Before release the tick below asks them every 10ms, because it is
193
+ * also polling for the release file; after release it used to stop asking
194
+ * entirely, leaving the provider's own exit as the gate's only remaining
195
+ * liveness check. A controller that died without cleaning up, or a run
196
+ * directory deleted underneath a live attempt, therefore left the provider
197
+ * running with nobody watching -- the stranded-process shape ADR 0010
198
+ * describes, in the one window the pre-release check does not cover.
199
+ *
200
+ * A second, not ten milliseconds: this watches a provider that runs for
201
+ * minutes, and three syscalls a second is the whole cost of never stranding
202
+ * one.
203
+ */
204
+ const WATCHDOG_INTERVAL_MS = 1_000;
205
+
189
206
  const timer = setInterval(() => {
190
207
  if (!parentAlive()) { clearInterval(timer); stopProvider(); return; }
191
208
  if (releaseDirectoryGone()) { clearInterval(timer); stopProvider(); return; }
192
209
  if (!existsSync(releasePath)) return;
193
210
  clearInterval(timer);
211
+ const watchdog = setInterval(() => {
212
+ if (parentAlive() && !releaseDirectoryGone()) return;
213
+ clearInterval(watchdog);
214
+ stopProvider();
215
+ }, WATCHDOG_INTERVAL_MS);
194
216
  const stdoutFd = openSync(config.stdoutPath, "wx", 0o600);
195
217
  const stderrFd = openSync(config.stderrPath, "wx", 0o600);
196
218
  const invocation = spawnInvocation(config.executable, config.args, { cwd: config.cwd });
@@ -207,6 +229,7 @@ const timer = setInterval(() => {
207
229
  }
208
230
  provider.once("error", () => process.exitCode = 127);
209
231
  provider.once("close", (code) => {
232
+ clearInterval(watchdog);
210
233
  capLog(config.stdoutPath, config.harness === "codex");
211
234
  capLog(config.stderrPath);
212
235
  process.exit(code ?? 1);
@@ -147,8 +147,19 @@ export function terminalErrorCode(state) {
147
147
  * `turn_limit` is here and not among the timeout codes below: a turn the CLI
148
148
  * stopped itself at `--max-turns` exits cleanly with no seal yet, and the
149
149
  * next dispatch seals its worktree as it does for any previous attempt.
150
+ *
151
+ * `incomplete_stream` is a stream that ended before its terminal envelope --
152
+ * Claude with no `result`, codex with no `turn.completed`, dsh with neither
153
+ * terminal record (`harnesses/protocol.mjs`). That is the provider's transport
154
+ * dying mid-turn, not the run's own doing, so it is the same class as
155
+ * `provider_error` and earns the same one retry. It was absent, and a worker
156
+ * whose provider dropped its connection parked on the first attempt while a
157
+ * worker whose provider returned an error envelope got a second one -- the
158
+ * harsher treatment for the less informative failure. The judge role is
159
+ * unaffected: `engine/settle-judge.mjs` routes this code to its own bounded
160
+ * re-ask before anything parks.
150
161
  */
151
- export const AUTO_RETRY_CODES = new Set(["judge_unavailable", "provider_error", "stall_timeout", "wall_clock_timeout", "turn_limit"]);
162
+ export const AUTO_RETRY_CODES = new Set(["judge_unavailable", "provider_error", "incomplete_stream", "stall_timeout", "wall_clock_timeout", "turn_limit"]);
152
163
 
153
164
  /**
154
165
  * Timeout codes earn their automatic retry only when phase 5b sealed work