faberun 0.19.1 → 0.19.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/faberun/references/contract.md +17 -5
- package/src/campaign/index.mjs +56 -1
- package/src/campaign/watch.mjs +197 -0
- package/src/cli/campaign.mjs +32 -180
- package/src/cli/launch.mjs +8 -2
- package/src/cli/plan.mjs +173 -27
- package/src/cli/spec.mjs +17 -7
- package/src/cli.mjs +2 -0
- package/src/contract/definition-of-done.mjs +50 -0
- package/src/contract/final-verification.mjs +21 -0
- package/src/contract/index.mjs +20 -8
- package/src/contract/verification.mjs +75 -3
- package/src/engine/dispatch.mjs +2 -1
- package/src/engine/gate.mjs +24 -1
- package/src/engine/lifecycle.mjs +12 -1
- package/src/engine/process.mjs +21 -1
- package/src/engine/scheduler.mjs +2 -2
- package/src/plan/pipeline.mjs +32 -5
- package/src/plan/proof-run.mjs +121 -0
- package/src/plan/repo-facts.mjs +5 -39
- package/src/plan/sizing.mjs +72 -4
- package/src/plan/spec.mjs +25 -1
- package/src/plan/template.mjs +24 -2
- package/src/report/final.mjs +44 -7
- package/src/report/render.mjs +98 -139
- package/src/report/role-usage.mjs +145 -0
package/src/cli/plan.mjs
CHANGED
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
* contested. This file only owns the wire — `src/plan/pipeline.mjs` owns the
|
|
5
5
|
* sequencing and every decision the pipeline makes.
|
|
6
6
|
*/
|
|
7
|
-
import { readFileSync } from "node:fs";
|
|
8
|
-
import { resolve } from "node:path";
|
|
7
|
+
import { mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
8
|
+
import { dirname, join, resolve } from "node:path";
|
|
9
9
|
import { detachArgv, detachSelf, waitForBootstrap } from "./launch.mjs";
|
|
10
10
|
import { classifyRunProgress } from "../campaign/chain.mjs";
|
|
11
11
|
import { runProgress } from "../engine/supervise.mjs";
|
|
@@ -15,11 +15,23 @@ import { validateFinalVerification, validateSharedVerification } from "../contra
|
|
|
15
15
|
import { colorLevel, statusToken } from "./brand.mjs";
|
|
16
16
|
import { delay } from "../util.mjs";
|
|
17
17
|
import { runPlanningPipeline } from "../plan/pipeline.mjs";
|
|
18
|
-
import { runDirectory } from "../run/paths.mjs";
|
|
18
|
+
import { campaignTree, runDirectory } from "../run/paths.mjs";
|
|
19
|
+
import { readCampaign } from "../campaign/record.mjs";
|
|
19
20
|
|
|
20
21
|
/** How often a foreground `plan` polls a launched stage's run directory. */
|
|
21
22
|
const DEFAULT_POLL_MS = 1_000;
|
|
22
23
|
|
|
24
|
+
/**
|
|
25
|
+
* How long a `--detach` launcher stays to see whether the planning process it
|
|
26
|
+
* started is actually up. Long enough to cover the bootstrap work that fails
|
|
27
|
+
* synchronously -- spec read, strict validation, catalogue load, the runtime
|
|
28
|
+
* ask -- and short enough that a launcher is not a supervisor: past this
|
|
29
|
+
* window, a planning run that dies is a running campaign's problem and leaves
|
|
30
|
+
* its evidence in the run directory, not here.
|
|
31
|
+
*/
|
|
32
|
+
const PLAN_BOOTSTRAP_WINDOW_MS = 5_000;
|
|
33
|
+
const PLAN_BOOTSTRAP_POLL_MS = 100;
|
|
34
|
+
|
|
23
35
|
/**
|
|
24
36
|
* `--runtime-defaults worker=<id>,judge=<id>`, either key optional, comma
|
|
25
37
|
* separated. Absent entirely, the pipeline falls through to plain
|
|
@@ -114,7 +126,7 @@ export function loadVerificationSuites(path) {
|
|
|
114
126
|
|
|
115
127
|
/**
|
|
116
128
|
* @param {string} target
|
|
117
|
-
* @param {{campaign?: string, phase?: string, "review-rounds"?: string, "approve-below"?: string, "runtime-defaults"?: string, runtimes?: string, verification?: string, detach?: boolean, json?: boolean}} values
|
|
129
|
+
* @param {{campaign?: string, phase?: string, "review-rounds"?: string, "approve-below"?: string, "runtime-defaults"?: string, runtimes?: string, verification?: string, package?: string, "targeted-fix"?: boolean, detach?: boolean, json?: boolean}} values
|
|
118
130
|
* @returns {Promise<void>}
|
|
119
131
|
*/
|
|
120
132
|
export async function planCli(target, values) {
|
|
@@ -131,6 +143,7 @@ export async function planCli(target, values) {
|
|
|
131
143
|
const verification = typeof values.verification === "string" && values.verification
|
|
132
144
|
? loadVerificationSuites(values.verification)
|
|
133
145
|
: {};
|
|
146
|
+
const packageMode = packageModeOf(values.package);
|
|
134
147
|
|
|
135
148
|
if (values.detach === true) {
|
|
136
149
|
const argv = ["plan", specPath, "--campaign", campaignId, "--phase", phase, "--review-rounds", String(reviewRounds)];
|
|
@@ -138,35 +151,71 @@ export async function planCli(target, values) {
|
|
|
138
151
|
if (values["runtime-defaults"] !== undefined) argv.push("--runtime-defaults", values["runtime-defaults"]);
|
|
139
152
|
if (typeof values.runtimes === "string" && values.runtimes) argv.push("--runtimes", resolve(values.runtimes));
|
|
140
153
|
if (typeof values.verification === "string" && values.verification) argv.push("--verification", resolve(values.verification));
|
|
154
|
+
if (packageMode !== "implementation") argv.push("--package", packageMode);
|
|
155
|
+
if (values["targeted-fix"] === true) argv.push("--targeted-fix");
|
|
156
|
+
const failurePath = planBootstrapFailurePath(process.cwd(), campaignId, phase);
|
|
157
|
+
// Read the campaign before creating anything: the failure record lives
|
|
158
|
+
// inside the campaign tree, so a typo in --campaign would otherwise leave
|
|
159
|
+
// a campaign directory with no record in it for `discoverCampaigns` to
|
|
160
|
+
// find. The child reads it too; this is the launcher refusing what it can
|
|
161
|
+
// see for itself rather than detaching into a certain failure.
|
|
162
|
+
readCampaign(campaignTree(process.cwd(), campaignId));
|
|
163
|
+
mkdirSync(dirname(failurePath), { recursive: true });
|
|
164
|
+
rmSync(failurePath, { force: true });
|
|
141
165
|
const child = detachArgv(argv);
|
|
142
166
|
if (child.pid === undefined) throw new Error("detached plan has no pid");
|
|
167
|
+
// The child's stdio is discarded (detachArgv), so a planning run that dies
|
|
168
|
+
// during bootstrap used to take its own reason with it while the launcher
|
|
169
|
+
// had already printed a pid and exited 0. Measured 2026-09-21 on macOS and
|
|
170
|
+
// Linux the same day: a run died after collecting repo facts and the stderr
|
|
171
|
+
// went with the closed connection. Waiting out a bounded window is the
|
|
172
|
+
// whole check -- a controller still alive past it is up, and one that is
|
|
173
|
+
// not has written why.
|
|
174
|
+
const failure = await watchPlanBootstrap(child, failurePath);
|
|
175
|
+
if (failure) {
|
|
176
|
+
process.stderr.write(`[plan] bootstrap failed · ${failure.error}\n[plan] recorded at ${failurePath}\n`);
|
|
177
|
+
process.exitCode = 1;
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
143
180
|
process.stdout.write(`[plan] detached · pid ${child.pid} · ${specPath}\n`);
|
|
144
181
|
return;
|
|
145
182
|
}
|
|
146
183
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
184
|
+
// A detached child arrives here with its stdio already discarded, so what
|
|
185
|
+
// it throws reaches nobody unless it is written down first. The record is
|
|
186
|
+
// written on every path, detached or not: a foreground failure that also
|
|
187
|
+
// leaves the file costs nothing and reads the same.
|
|
188
|
+
let result;
|
|
189
|
+
try {
|
|
190
|
+
result = await runPlanningPipeline({
|
|
191
|
+
specPath,
|
|
192
|
+
campaignId,
|
|
193
|
+
phase,
|
|
194
|
+
reviewRounds,
|
|
195
|
+
approveBelow,
|
|
196
|
+
runtimeDefaults,
|
|
197
|
+
runtimes,
|
|
198
|
+
verification,
|
|
199
|
+
targetedFix: values["targeted-fix"] === true,
|
|
200
|
+
packageMode,
|
|
201
|
+
launch: async (contractPath, contract) => {
|
|
202
|
+
const child = detachSelf("run", contractPath);
|
|
203
|
+
if (child.pid === undefined) throw new Error("detached planning run has no pid");
|
|
204
|
+
await waitForBootstrap(runDirectory(contract.cwd, contract.id), child.pid, child);
|
|
205
|
+
},
|
|
206
|
+
wait: async (runDir) => {
|
|
207
|
+
for (;;) {
|
|
208
|
+
const progress = runProgress(runDir);
|
|
209
|
+
const classification = classifyRunProgress(progress);
|
|
210
|
+
if (classification !== "unfinished" && classification !== "waiting") return progress;
|
|
211
|
+
await delay(DEFAULT_POLL_MS);
|
|
212
|
+
}
|
|
213
|
+
},
|
|
214
|
+
});
|
|
215
|
+
} catch (error) {
|
|
216
|
+
writePlanBootstrapFailure(process.cwd(), campaignId, phase, error instanceof Error ? error : new Error(String(error)));
|
|
217
|
+
throw error;
|
|
218
|
+
}
|
|
170
219
|
|
|
171
220
|
if (values.json === true) {
|
|
172
221
|
process.stdout.write(`${JSON.stringify(result)}\n`);
|
|
@@ -180,3 +229,100 @@ export async function planCli(target, values) {
|
|
|
180
229
|
for (const warning of result.warnings) process.stdout.write(`${statusToken("warn", colorLevel(process.env, process.stdout.isTTY))} ${warning}\n`);
|
|
181
230
|
process.stdout.write(`[plan] ${campaignId} phase ${phase} frozen · approved ${result.approved} · ${result.contractPath}\n`);
|
|
182
231
|
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Where a detached planning run records why it never came up. It sits beside
|
|
235
|
+
* the phase's durable plan artifacts rather than in the disposable scratch
|
|
236
|
+
* tree, and it is derived from campaign and phase alone -- the launcher and
|
|
237
|
+
* the child compute the same path without either having to parse the spec.
|
|
238
|
+
*
|
|
239
|
+
* @param {string} cwd
|
|
240
|
+
* @param {string} campaignId
|
|
241
|
+
* @param {string} phase
|
|
242
|
+
* @returns {string}
|
|
243
|
+
*/
|
|
244
|
+
export function planBootstrapFailurePath(cwd, campaignId, phase) {
|
|
245
|
+
return join(campaignTree(cwd, campaignId), "plans", phase, "bootstrap-failure.json");
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Record a detached planning run's bootstrap failure where its launcher can
|
|
250
|
+
* read it. Best effort: a failure to write this must never replace the
|
|
251
|
+
* failure it was describing.
|
|
252
|
+
*
|
|
253
|
+
* @param {string} cwd
|
|
254
|
+
* @param {string} campaignId
|
|
255
|
+
* @param {string} phase
|
|
256
|
+
* @param {Error} error
|
|
257
|
+
* @returns {void}
|
|
258
|
+
*/
|
|
259
|
+
export function writePlanBootstrapFailure(cwd, campaignId, phase, error) {
|
|
260
|
+
try {
|
|
261
|
+
const path = planBootstrapFailurePath(cwd, campaignId, phase);
|
|
262
|
+
mkdirSync(dirname(path), { recursive: true });
|
|
263
|
+
writeFileSync(path, `${JSON.stringify({ at: new Date().toISOString(), pid: process.pid, campaignId, phase, error: error.message }, null, 2)}\n`);
|
|
264
|
+
} catch {
|
|
265
|
+
// Nothing left to do: the caller is already reporting the real failure.
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Watch a freshly detached planning child through its bootstrap window.
|
|
271
|
+
*
|
|
272
|
+
* Returns the recorded failure when the child died inside the window, and
|
|
273
|
+
* null when it is still running at the end of it. A child that exits zero
|
|
274
|
+
* inside the window also reads as no failure: a planning run can legitimately
|
|
275
|
+
* be that fast only by refusing early, and it will have written its own
|
|
276
|
+
* record if it refused.
|
|
277
|
+
*
|
|
278
|
+
* @param {import("node:child_process").ChildProcess} child
|
|
279
|
+
* @param {string} failurePath
|
|
280
|
+
* @param {{windowMs?: number, pollMs?: number}} [options]
|
|
281
|
+
* @returns {Promise<{error: string}|null>}
|
|
282
|
+
*/
|
|
283
|
+
export async function watchPlanBootstrap(child, failurePath, { windowMs = PLAN_BOOTSTRAP_WINDOW_MS, pollMs = PLAN_BOOTSTRAP_POLL_MS } = {}) {
|
|
284
|
+
let exitCode = /** @type {number|null|undefined} */ (undefined);
|
|
285
|
+
let exited = false;
|
|
286
|
+
child.once("exit", (code) => { exited = true; exitCode = code; });
|
|
287
|
+
const deadline = Date.now() + windowMs;
|
|
288
|
+
while (Date.now() < deadline) {
|
|
289
|
+
// Typed checks, not `!== null`: a child object that never carried the
|
|
290
|
+
// property at all would otherwise read as one that has already exited.
|
|
291
|
+
if (exited || typeof child.exitCode === "number" || typeof child.signalCode === "string") {
|
|
292
|
+
const recorded = readPlanBootstrapFailure(failurePath);
|
|
293
|
+
if (recorded) return recorded;
|
|
294
|
+
const code = typeof exitCode === "number" ? exitCode : child.exitCode;
|
|
295
|
+
if (typeof code === "number" && code !== 0) return { error: `the detached planning process exited ${code} without recording a reason` };
|
|
296
|
+
const signal = child.signalCode;
|
|
297
|
+
if (typeof signal === "string") return { error: `the detached planning process was killed by ${signal} without recording a reason` };
|
|
298
|
+
return null;
|
|
299
|
+
}
|
|
300
|
+
await delay(pollMs);
|
|
301
|
+
}
|
|
302
|
+
return null;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
/** @param {string} failurePath @returns {{error: string}|null} */
|
|
306
|
+
function readPlanBootstrapFailure(failurePath) {
|
|
307
|
+
try {
|
|
308
|
+
const record = JSON.parse(readFileSync(failurePath, "utf8"));
|
|
309
|
+
return typeof record?.error === "string" ? { error: record.error } : null;
|
|
310
|
+
} catch {
|
|
311
|
+
return null;
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/**
|
|
316
|
+
* `--package implementation|exploratory`. Implementation is the default and
|
|
317
|
+
* the only mode there was: nodes sized by their write set. Exploratory sizes
|
|
318
|
+
* by what a node reads, accepts a one-file write set as the normal shape of a
|
|
319
|
+
* finding, and reports a node whose read surface dwarfs its siblings'.
|
|
320
|
+
*
|
|
321
|
+
* @param {unknown} value
|
|
322
|
+
* @returns {import("../plan/sizing.mjs").PackageMode}
|
|
323
|
+
*/
|
|
324
|
+
export function packageModeOf(value) {
|
|
325
|
+
if (value === undefined) return "implementation";
|
|
326
|
+
if (value === "implementation" || value === "exploratory") return value;
|
|
327
|
+
throw new Error(`--package must be implementation or exploratory: ${String(value)}`);
|
|
328
|
+
}
|
package/src/cli/spec.mjs
CHANGED
|
@@ -14,7 +14,7 @@ import { validateSpec } from "../plan/spec.mjs";
|
|
|
14
14
|
/** Flags are scoped to the operation that declares them; all others are rejected. */
|
|
15
15
|
/** @type {Record<string, import("node:util").ParseArgsOptionsConfig>} */
|
|
16
16
|
const OPERATION_OPTIONS = {
|
|
17
|
-
validate: { "strict-traceability": { type: "boolean" }, json: { type: "boolean" } },
|
|
17
|
+
validate: { "strict-traceability": { type: "boolean" }, "run-proofs": { type: "boolean" }, json: { type: "boolean" } },
|
|
18
18
|
scaffold: { id: { type: "string" } },
|
|
19
19
|
};
|
|
20
20
|
|
|
@@ -63,9 +63,13 @@ export function specCli(args) {
|
|
|
63
63
|
}
|
|
64
64
|
const target = parsed.positionals[0];
|
|
65
65
|
if (!target || parsed.positionals.length > 1) return usage();
|
|
66
|
-
const values = /** @type {{"strict-traceability"?: boolean, json?: boolean, id?: string}} */ (parsed.values);
|
|
66
|
+
const values = /** @type {{"strict-traceability"?: boolean, "run-proofs"?: boolean, json?: boolean, id?: string}} */ (parsed.values);
|
|
67
67
|
if (operation === "validate") {
|
|
68
|
-
validateSpecFile(resolve(target), {
|
|
68
|
+
validateSpecFile(resolve(target), {
|
|
69
|
+
strict: values["strict-traceability"] === true,
|
|
70
|
+
runProofs: values["run-proofs"] === true,
|
|
71
|
+
json: values.json === true,
|
|
72
|
+
});
|
|
69
73
|
return;
|
|
70
74
|
}
|
|
71
75
|
try {
|
|
@@ -80,12 +84,18 @@ export function specCli(args) {
|
|
|
80
84
|
* Validate a spec file and print its class, its overall verdict, and one
|
|
81
85
|
* line per finding. Exits `1` when the verdict is not `ok`.
|
|
82
86
|
*
|
|
87
|
+
* `runProofs` is the only operation here that spawns anything: it runs each
|
|
88
|
+
* requirement's declared proof instead of only checking that one is written
|
|
89
|
+
* down. It is opt-in because running a repository's proofs costs real time,
|
|
90
|
+
* and default-off keeps `spec validate` the deterministic, side-effect-free
|
|
91
|
+
* read it has always been.
|
|
92
|
+
*
|
|
83
93
|
* @param {string} path
|
|
84
|
-
* @param {{strict: boolean, json: boolean}} options
|
|
94
|
+
* @param {{strict: boolean, json: boolean, runProofs?: boolean}} options
|
|
85
95
|
* @returns {SpecValidation}
|
|
86
96
|
*/
|
|
87
|
-
export function validateSpecFile(path, { strict, json }) {
|
|
88
|
-
const result = validateSpec(readFileSync(path, "utf8"), { cwd: process.cwd(), strict });
|
|
97
|
+
export function validateSpecFile(path, { strict, json, runProofs = false }) {
|
|
98
|
+
const result = validateSpec(readFileSync(path, "utf8"), { cwd: process.cwd(), strict, runProofs });
|
|
89
99
|
if (json) {
|
|
90
100
|
process.stdout.write(`${JSON.stringify(result)}\n`);
|
|
91
101
|
} else {
|
|
@@ -112,7 +122,7 @@ export function scaffoldSpec(path, id) {
|
|
|
112
122
|
|
|
113
123
|
/** @returns {void} */
|
|
114
124
|
function usage() {
|
|
115
|
-
process.stderr.write("usage: faberun spec <validate|scaffold> <path> [--strict-traceability] [--json] [--id <value>]\n");
|
|
125
|
+
process.stderr.write("usage: faberun spec <validate|scaffold> <path> [--strict-traceability] [--run-proofs] [--json] [--id <value>]\n");
|
|
116
126
|
process.exitCode = 2;
|
|
117
127
|
}
|
|
118
128
|
|
package/src/cli.mjs
CHANGED
|
@@ -123,6 +123,8 @@ export const COMMAND_OPTIONS = {
|
|
|
123
123
|
"runtime-defaults": { type: "string" },
|
|
124
124
|
runtimes: { type: "string" },
|
|
125
125
|
verification: { type: "string" },
|
|
126
|
+
package: { type: "string" },
|
|
127
|
+
"targeted-fix": { type: "boolean" },
|
|
126
128
|
detach: { type: "boolean" },
|
|
127
129
|
json: { type: "boolean" },
|
|
128
130
|
},
|
|
@@ -95,3 +95,53 @@ function validateVerificationRef(value, label, options) {
|
|
|
95
95
|
return String(index);
|
|
96
96
|
}
|
|
97
97
|
|
|
98
|
+
/**
|
|
99
|
+
* A `kind: "command"` proof's `ref` runs through a shell (`judge-gate.mjs`'s
|
|
100
|
+
* `proveCommand`), while a task packet's `verification` entries run as argv
|
|
101
|
+
* arrays -- the opposite quoting convention for the same author intent.
|
|
102
|
+
* Measured 2026-09-22: six DoD refs on one campaign wrote
|
|
103
|
+
* `--test-name-pattern=a b c` the way an argv array would take it, the shell
|
|
104
|
+
* split it into three words, and the gate rejected work whose identical
|
|
105
|
+
* command had just passed as verification. Nothing warned the author, because
|
|
106
|
+
* a shell that receives extra bare words after an unquoted flag value does not
|
|
107
|
+
* itself know they were meant to be one argument.
|
|
108
|
+
*
|
|
109
|
+
* This flags the same shape rather than every proof: a node:test filter flag
|
|
110
|
+
* (the flags `declaredTestFilters` in judge-gate.mjs recognizes) whose
|
|
111
|
+
* unquoted value is immediately followed by bare words is the pattern the
|
|
112
|
+
* incident measured, and it is precise enough that a legitimate `ref` rarely
|
|
113
|
+
* has trailing bare words right after such a flag by accident.
|
|
114
|
+
*
|
|
115
|
+
* @param {DefinitionOfDoneItem[]} items
|
|
116
|
+
* @param {number} index
|
|
117
|
+
* @returns {string[]}
|
|
118
|
+
*/
|
|
119
|
+
export function unquotedFilterValueWarnings(items, index) {
|
|
120
|
+
/** @type {string[]} */
|
|
121
|
+
const warnings = [];
|
|
122
|
+
items.forEach((item, itemIndex) => {
|
|
123
|
+
if (item.proof?.kind !== "command") return;
|
|
124
|
+
const tokens = item.proof.ref.split(/\s+/u).filter(Boolean);
|
|
125
|
+
for (const flag of TEST_FILTER_FLAGS) {
|
|
126
|
+
for (let position = 0; position < tokens.length; position++) {
|
|
127
|
+
const prefix = `${flag}=`;
|
|
128
|
+
if (!tokens[position].startsWith(prefix)) continue;
|
|
129
|
+
const value = tokens[position].slice(prefix.length);
|
|
130
|
+
if (/^['"]/u.test(value)) continue;
|
|
131
|
+
let end = position + 1;
|
|
132
|
+
while (end < tokens.length && !tokens[end].startsWith("-")) end++;
|
|
133
|
+
if (end === position + 1) continue;
|
|
134
|
+
const spilled = [value, ...tokens.slice(position + 1, end)].join(" ");
|
|
135
|
+
warnings.push(
|
|
136
|
+
`nodes[${index}] (definitionOfDone[${itemIndex}]): proof.ref's ${flag} value "${spilled}" is unquoted; ` +
|
|
137
|
+
`kind: "command" runs through a shell, unlike taskPacket.verification's argv, so the space splits it into extra words -- quote it as ${flag}="${spilled}"`,
|
|
138
|
+
);
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
});
|
|
142
|
+
return warnings;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** The node:test filter flags a `kind: "command"` proof's shell can split on an unquoted value. */
|
|
146
|
+
const TEST_FILTER_FLAGS = ["--test-name-pattern", "--test-skip-pattern"];
|
|
147
|
+
|
|
@@ -60,6 +60,27 @@ export function sharedVerificationCommands(contract) {
|
|
|
60
60
|
return contract.sharedVerification ?? [];
|
|
61
61
|
}
|
|
62
62
|
|
|
63
|
+
/**
|
|
64
|
+
* The timeout a Definition of Done `command` proof is spawned with. It used
|
|
65
|
+
* to be capped at a hardcoded 120s regardless of what the node's own attempt
|
|
66
|
+
* was given, so a contract that raised `timeoutSec` to run a slow command as
|
|
67
|
+
* `taskPacket.verification` still had the identical command fail its DoD
|
|
68
|
+
* proof on the same run: measured 2026-09-22, six proofs rejected work whose
|
|
69
|
+
* argv had just passed with a `timeoutSec` well past 120s. A command proof
|
|
70
|
+
* gets the same budget the node's own attempt has, because that is the
|
|
71
|
+
* budget the author already reasoned about; nothing here invents a second
|
|
72
|
+
* number for the gate to disagree with. Shared by `dispatch.mjs` (the actual
|
|
73
|
+
* spawn) and `scheduler.mjs` (the freeze-detection budget that must not judge
|
|
74
|
+
* a node frozen before its own gate's timeout has had a chance to fire).
|
|
75
|
+
*
|
|
76
|
+
* @param {{timeoutSec?: number}} node
|
|
77
|
+
* @param {{timeoutSec?: number}} contract
|
|
78
|
+
* @returns {number}
|
|
79
|
+
*/
|
|
80
|
+
export function gateProofTimeoutMs(node, contract) {
|
|
81
|
+
return Math.max(1_000, (node.timeoutSec ?? contract.timeoutSec ?? 60) * 1_000);
|
|
82
|
+
}
|
|
83
|
+
|
|
63
84
|
/**
|
|
64
85
|
* Whether no other node in the contract depends on this one. `finalVerification`
|
|
65
86
|
* is a candidate to run on a phase-terminal node only; a node with a dependant
|
package/src/contract/index.mjs
CHANGED
|
@@ -3,9 +3,9 @@ import { readFileSync, statSync } from "node:fs";
|
|
|
3
3
|
import { dirname, resolve } from "node:path";
|
|
4
4
|
import { loadTaskPacket, renderWorkerPrompt } from "./task-packet.mjs";
|
|
5
5
|
import { RESERVED_ARTICLES } from "./articles.mjs";
|
|
6
|
-
import { validateDefinitionOfDone } from "./definition-of-done.mjs";
|
|
6
|
+
import { unquotedFilterValueWarnings, validateDefinitionOfDone } from "./definition-of-done.mjs";
|
|
7
7
|
import { validateFinalVerification, validateSharedVerification } from "./final-verification.mjs";
|
|
8
|
-
import { VERIFICATION_LIMITS } from "./verification.mjs";
|
|
8
|
+
import { VERIFICATION_LIMITS, requirementProofWarnings } from "./verification.mjs";
|
|
9
9
|
import {
|
|
10
10
|
validateCapabilityRequirements,
|
|
11
11
|
} from "../harnesses/index.mjs";
|
|
@@ -374,12 +374,24 @@ export function validateContract(raw, contractPath, options = {}) {
|
|
|
374
374
|
const sharedVerification = validateSharedVerification(raw.sharedVerification, "contract.sharedVerification");
|
|
375
375
|
const contractCommands = [...finalVerification ?? [], ...sharedVerification ?? []].map((command) => command.argv.join(" "));
|
|
376
376
|
const contractWrites = new Set(nodes.flatMap((node) => node.taskPacket.writeFiles ?? []));
|
|
377
|
-
const warnings =
|
|
378
|
-
...
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
377
|
+
const warnings = [
|
|
378
|
+
...nodes.flatMap((node, index) => [
|
|
379
|
+
...commandCoverageWarnings(node, index),
|
|
380
|
+
...unquotedFilterValueWarnings(node.definitionOfDone ?? [], index),
|
|
381
|
+
...(persisted ? [] : mirrorCoverageWarnings(node, index, cwd, contractCommands, contractWrites)),
|
|
382
|
+
...(persisted ? [] : unsnapshottedWriteWarnings(node, index, cwd)),
|
|
383
|
+
...(persisted ? [] : writeFileLineBudgetWarnings(node, index, cwd)),
|
|
384
|
+
]),
|
|
385
|
+
// Cross-node by construction: a requirement proven in two nodes is only
|
|
386
|
+
// visible when every node's commands are read together, which is the
|
|
387
|
+
// whole point -- one copy repaired and six left behind is what a per-node
|
|
388
|
+
// read cannot see.
|
|
389
|
+
...requirementProofWarnings([
|
|
390
|
+
...nodes.map((node, index) => ({ id: `nodes[${index}] (${node.id})`, requirementIds: node.requirementIds, commands: node.taskPacket.verification ?? [] })),
|
|
391
|
+
{ id: "contract.sharedVerification", commands: sharedVerification ?? [] },
|
|
392
|
+
{ id: "contract.finalVerification", commands: finalVerification ?? [] },
|
|
393
|
+
]),
|
|
394
|
+
];
|
|
383
395
|
const contract = /** @type {ValidatedContract} */ ({
|
|
384
396
|
...raw,
|
|
385
397
|
schemaVersion: /** @type {number} */ (raw.schemaVersion),
|
|
@@ -8,7 +8,18 @@ export const VERIFICATION_LIMITS = Object.freeze({
|
|
|
8
8
|
stderrBytes: 16 * 1024,
|
|
9
9
|
maxCommands: 32,
|
|
10
10
|
maxRepeat: 8,
|
|
11
|
-
|
|
11
|
+
// A verification command may declare up to half an hour. It was 600s, which
|
|
12
|
+
// is below this repository's own suite: measured 2026-09-22, `npm test`
|
|
13
|
+
// takes 434-473s at the default parallelism on the author's machine and 650s
|
|
14
|
+
// on a Windows CI runner, and `node --test --test-concurrency=1 test/engine/`
|
|
15
|
+
// -- the way a packet's verification actually runs it -- takes 1035-1058s.
|
|
16
|
+
// So the one command that proves the engine could not be declared at all,
|
|
17
|
+
// and a packet author's way out was `--test-name-pattern`, which exits 0
|
|
18
|
+
// when it matches nothing (see AGENTS.md). A cap that pushes authors toward
|
|
19
|
+
// a proof that certifies nothing is worse than a longer runaway. The node's
|
|
20
|
+
// own wall clock (`contract.timeoutSec`, 2400s by default) still bounds the
|
|
21
|
+
// attempt above this.
|
|
22
|
+
maxTimeoutSec: 1_800,
|
|
12
23
|
stateStdoutBytes: 2 * 1024,
|
|
13
24
|
stateCommands: 16,
|
|
14
25
|
stateAttempts: 4,
|
|
@@ -42,7 +53,14 @@ export const MUTATION_TIERS = Object.freeze({
|
|
|
42
53
|
/**
|
|
43
54
|
* One declared deterministic check: an argv command run by the controller.
|
|
44
55
|
*
|
|
45
|
-
*
|
|
56
|
+
* `requirementId` names the spec requirement this command is the proof of. It
|
|
57
|
+
* changes nothing about how the command runs; it is what makes a duplicated
|
|
58
|
+
* proof visible. Measured 2026-09-22: one broken command lived in a spec's R3,
|
|
59
|
+
* in its R4 and in seven nodes' verification, and the repair reached one of
|
|
60
|
+
* them — nothing could tell that the other copies had stopped agreeing,
|
|
61
|
+
* because nothing recorded that they were copies of one claim.
|
|
62
|
+
*
|
|
63
|
+
* @typedef {{argv: string[], cwd?: string, timeoutSec?: number, repeat?: number, env?: string[], mutation?: {tier: MutationTier}, requirementId?: string}} VerificationCommand
|
|
46
64
|
*/
|
|
47
65
|
|
|
48
66
|
/**
|
|
@@ -122,7 +140,7 @@ export function validateVerificationCommands(commands, label = "verification") {
|
|
|
122
140
|
function validateVerificationCommand(command, label = "verification command") {
|
|
123
141
|
if (!command || typeof command !== "object" || Array.isArray(command)) throw new TypeError(`${label} must be an argv command object`);
|
|
124
142
|
const record = /** @type {Record<string, unknown>} */ (command);
|
|
125
|
-
const allowed = new Set(["argv", "cwd", "timeoutSec", "repeat", "env", "mutation"]);
|
|
143
|
+
const allowed = new Set(["argv", "cwd", "timeoutSec", "repeat", "env", "mutation", "requirementId"]);
|
|
126
144
|
for (const key of Object.keys(record)) if (!allowed.has(key)) throw new TypeError(`${label} has unexpected field ${key}`);
|
|
127
145
|
if (!Array.isArray(record.argv) || record.argv.length === 0 || record.argv.length > 64 || record.argv.some((item) => typeof item !== "string" || !item.trim() || Buffer.byteLength(item, "utf8") > 8 * 1024)) {
|
|
128
146
|
throw new TypeError(`${label}.argv must be a non-empty array of strings`);
|
|
@@ -157,10 +175,18 @@ function validateVerificationCommand(command, label = "verification command") {
|
|
|
157
175
|
}
|
|
158
176
|
mutation = { tier: /** @type {MutationTier} */ (mutationRecord.tier) };
|
|
159
177
|
}
|
|
178
|
+
// Bounded exactly like the node-level `requirementIds` it must match against
|
|
179
|
+
// (at most 128 bytes), so the two sides of the claim cannot accept different
|
|
180
|
+
// ids.
|
|
181
|
+
if (record.requirementId !== undefined
|
|
182
|
+
&& (typeof record.requirementId !== "string" || !record.requirementId.trim() || Buffer.byteLength(record.requirementId, "utf8") > 128)) {
|
|
183
|
+
throw new TypeError(`${label}.requirementId must be a requirement id of at most 128 bytes`);
|
|
184
|
+
}
|
|
160
185
|
/** @type {VerificationCommand} */
|
|
161
186
|
const normalized = { argv: [.../** @type {string[]} */ (record.argv)], timeoutSec, repeat, env: [.../** @type {string[]} */ (env)] };
|
|
162
187
|
if (record.cwd !== undefined) normalized.cwd = /** @type {string} */ (record.cwd);
|
|
163
188
|
if (mutation !== undefined) normalized.mutation = mutation;
|
|
189
|
+
if (record.requirementId !== undefined) normalized.requirementId = /** @type {string} */ (record.requirementId);
|
|
164
190
|
return normalized;
|
|
165
191
|
}
|
|
166
192
|
|
|
@@ -199,3 +225,49 @@ export function compactVerification(result) {
|
|
|
199
225
|
return { passed: Boolean(result?.passed), commands };
|
|
200
226
|
}
|
|
201
227
|
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* One proof, one source. A command that declares `requirementId` says it is
|
|
231
|
+
* the proof of that requirement; this reports the two ways such a claim can
|
|
232
|
+
* be false.
|
|
233
|
+
*
|
|
234
|
+
* A claim the node does not carry is a mislabel: the node's own
|
|
235
|
+
* `requirementIds` are what the phase assigned it, and a command proving
|
|
236
|
+
* something outside them is either the wrong id or the wrong node.
|
|
237
|
+
*
|
|
238
|
+
* Copies that stopped agreeing are the measured one. 2026-09-22: a broken
|
|
239
|
+
* command lived in a spec's R3, its R4, and seven nodes' verification, and
|
|
240
|
+
* the repair reached one copy. Nothing could see that the others had drifted,
|
|
241
|
+
* because nothing recorded that they were copies of a single claim. Argv is
|
|
242
|
+
* compared against argv, never a joined string against a shell command: a
|
|
243
|
+
* joined argv loses argument boundaries, which is the same reason a
|
|
244
|
+
* `verification` proof references an index instead of comparing text.
|
|
245
|
+
*
|
|
246
|
+
* @param {Array<{id: string, requirementIds?: string[], commands: VerificationCommand[]}>} owners
|
|
247
|
+
* @returns {string[]}
|
|
248
|
+
*/
|
|
249
|
+
export function requirementProofWarnings(owners) {
|
|
250
|
+
/** @type {string[]} */
|
|
251
|
+
const warnings = [];
|
|
252
|
+
/** @type {Map<string, Array<{owner: string, position: number, argv: string[]}>>} */
|
|
253
|
+
const claims = new Map();
|
|
254
|
+
for (const owner of owners) {
|
|
255
|
+
owner.commands.forEach((command, position) => {
|
|
256
|
+
const requirementId = command.requirementId;
|
|
257
|
+
if (requirementId === undefined) return;
|
|
258
|
+
if (owner.requirementIds !== undefined && !owner.requirementIds.includes(requirementId)) {
|
|
259
|
+
warnings.push(`${owner.id}: verification[${position}] declares requirementId "${requirementId}", which this node does not carry in requirementIds`);
|
|
260
|
+
}
|
|
261
|
+
const claimed = claims.get(requirementId) ?? [];
|
|
262
|
+
claimed.push({ owner: owner.id, position, argv: command.argv });
|
|
263
|
+
claims.set(requirementId, claimed);
|
|
264
|
+
});
|
|
265
|
+
}
|
|
266
|
+
for (const [requirementId, claimed] of claims) {
|
|
267
|
+
const distinct = new Map(claimed.map((claim) => [JSON.stringify(claim.argv), claim]));
|
|
268
|
+
if (distinct.size < 2) continue;
|
|
269
|
+
const listed = [...distinct.values()].map((claim) => `${claim.owner}: verification[${claim.position}] runs ${JSON.stringify(claim.argv)}`).join("; ");
|
|
270
|
+
warnings.push(`requirementId "${requirementId}" is proven by commands that no longer agree: ${listed}`);
|
|
271
|
+
}
|
|
272
|
+
return warnings;
|
|
273
|
+
}
|
package/src/engine/dispatch.mjs
CHANGED
|
@@ -25,6 +25,7 @@ import { attemptWorkspace, createAttemptWorktree, sealAttempt } from "../repo/wo
|
|
|
25
25
|
import { attemptWorktreePath } from "../run/paths.mjs";
|
|
26
26
|
import { basename, dirname, join } from "node:path";
|
|
27
27
|
import { errorCode, errorMessage } from "../util.mjs";
|
|
28
|
+
import { gateProofTimeoutMs } from "../contract/final-verification.mjs";
|
|
28
29
|
import { captureWorkspaceScope, captureWorkspaceSnapshot } from "../repo/workspace.mjs";
|
|
29
30
|
import { deterministicGate, judgeReaskReason, judgeRequired, judgeSkippedByScope } from "./judge-gate.mjs";
|
|
30
31
|
import { emptyScope, persistedScopeBoundary, workerScope } from "./scope.mjs";
|
|
@@ -459,7 +460,7 @@ export async function startJudge(contract, node, state, runDir, running, workerR
|
|
|
459
460
|
node,
|
|
460
461
|
workspace,
|
|
461
462
|
reask,
|
|
462
|
-
|
|
463
|
+
gateProofTimeoutMs(node, contract),
|
|
463
464
|
/** @type {import("../contract/index.mjs").VerificationState|null} */ (state.verification),
|
|
464
465
|
);
|
|
465
466
|
state.review = reviewMode(node.gate);
|
package/src/engine/gate.mjs
CHANGED
|
@@ -25,7 +25,8 @@
|
|
|
25
25
|
* It also exits when the release file's directory is gone (measured
|
|
26
26
|
* 2026-09-16: gate processes from a prior day's test runs, spawned into a
|
|
27
27
|
* temp directory the failed test never cleaned up, were still alive and
|
|
28
|
-
* waiting for a release file that could now never appear)
|
|
28
|
+
* waiting for a release file that could now never appear), and it keeps
|
|
29
|
+
* asking both of those questions after the provider starts, not only before.
|
|
29
30
|
*/
|
|
30
31
|
import { existsSync, readFileSync, statSync, openSync, closeSync, readSync, writeSync } from "node:fs";
|
|
31
32
|
import { dirname } from "node:path";
|
|
@@ -186,11 +187,32 @@ function releaseDirectoryGone() {
|
|
|
186
187
|
return !existsSync(dirname(releasePath));
|
|
187
188
|
}
|
|
188
189
|
|
|
190
|
+
/**
|
|
191
|
+
* How often the gate re-asks its two liveness questions once the provider is
|
|
192
|
+
* running. Before release the tick below asks them every 10ms, because it is
|
|
193
|
+
* also polling for the release file; after release it used to stop asking
|
|
194
|
+
* entirely, leaving the provider's own exit as the gate's only remaining
|
|
195
|
+
* liveness check. A controller that died without cleaning up, or a run
|
|
196
|
+
* directory deleted underneath a live attempt, therefore left the provider
|
|
197
|
+
* running with nobody watching -- the stranded-process shape ADR 0010
|
|
198
|
+
* describes, in the one window the pre-release check does not cover.
|
|
199
|
+
*
|
|
200
|
+
* A second, not ten milliseconds: this watches a provider that runs for
|
|
201
|
+
* minutes, and three syscalls a second is the whole cost of never stranding
|
|
202
|
+
* one.
|
|
203
|
+
*/
|
|
204
|
+
const WATCHDOG_INTERVAL_MS = 1_000;
|
|
205
|
+
|
|
189
206
|
const timer = setInterval(() => {
|
|
190
207
|
if (!parentAlive()) { clearInterval(timer); stopProvider(); return; }
|
|
191
208
|
if (releaseDirectoryGone()) { clearInterval(timer); stopProvider(); return; }
|
|
192
209
|
if (!existsSync(releasePath)) return;
|
|
193
210
|
clearInterval(timer);
|
|
211
|
+
const watchdog = setInterval(() => {
|
|
212
|
+
if (parentAlive() && !releaseDirectoryGone()) return;
|
|
213
|
+
clearInterval(watchdog);
|
|
214
|
+
stopProvider();
|
|
215
|
+
}, WATCHDOG_INTERVAL_MS);
|
|
194
216
|
const stdoutFd = openSync(config.stdoutPath, "wx", 0o600);
|
|
195
217
|
const stderrFd = openSync(config.stderrPath, "wx", 0o600);
|
|
196
218
|
const invocation = spawnInvocation(config.executable, config.args, { cwd: config.cwd });
|
|
@@ -207,6 +229,7 @@ const timer = setInterval(() => {
|
|
|
207
229
|
}
|
|
208
230
|
provider.once("error", () => process.exitCode = 127);
|
|
209
231
|
provider.once("close", (code) => {
|
|
232
|
+
clearInterval(watchdog);
|
|
210
233
|
capLog(config.stdoutPath, config.harness === "codex");
|
|
211
234
|
capLog(config.stderrPath);
|
|
212
235
|
process.exit(code ?? 1);
|
package/src/engine/lifecycle.mjs
CHANGED
|
@@ -147,8 +147,19 @@ export function terminalErrorCode(state) {
|
|
|
147
147
|
* `turn_limit` is here and not among the timeout codes below: a turn the CLI
|
|
148
148
|
* stopped itself at `--max-turns` exits cleanly with no seal yet, and the
|
|
149
149
|
* next dispatch seals its worktree as it does for any previous attempt.
|
|
150
|
+
*
|
|
151
|
+
* `incomplete_stream` is a stream that ended before its terminal envelope --
|
|
152
|
+
* Claude with no `result`, codex with no `turn.completed`, dsh with neither
|
|
153
|
+
* terminal record (`harnesses/protocol.mjs`). That is the provider's transport
|
|
154
|
+
* dying mid-turn, not the run's own doing, so it is the same class as
|
|
155
|
+
* `provider_error` and earns the same one retry. It was absent, and a worker
|
|
156
|
+
* whose provider dropped its connection parked on the first attempt while a
|
|
157
|
+
* worker whose provider returned an error envelope got a second one -- the
|
|
158
|
+
* harsher treatment for the less informative failure. The judge role is
|
|
159
|
+
* unaffected: `engine/settle-judge.mjs` routes this code to its own bounded
|
|
160
|
+
* re-ask before anything parks.
|
|
150
161
|
*/
|
|
151
|
-
export const AUTO_RETRY_CODES = new Set(["judge_unavailable", "provider_error", "stall_timeout", "wall_clock_timeout", "turn_limit"]);
|
|
162
|
+
export const AUTO_RETRY_CODES = new Set(["judge_unavailable", "provider_error", "incomplete_stream", "stall_timeout", "wall_clock_timeout", "turn_limit"]);
|
|
152
163
|
|
|
153
164
|
/**
|
|
154
165
|
* Timeout codes earn their automatic retry only when phase 5b sealed work
|