@wildorder/nightshift 0.5.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -21
- package/dist/author.d.ts +13 -0
- package/dist/author.d.ts.map +1 -1
- package/dist/author.js +46 -13
- package/dist/author.js.map +1 -1
- package/dist/ci-init.d.ts +51 -0
- package/dist/ci-init.d.ts.map +1 -0
- package/dist/ci-init.js +274 -0
- package/dist/ci-init.js.map +1 -0
- package/dist/cli.js +99 -11
- package/dist/cli.js.map +1 -1
- package/dist/decide.d.ts +13 -5
- package/dist/decide.d.ts.map +1 -1
- package/dist/decide.js +50 -35
- package/dist/decide.js.map +1 -1
- package/dist/decider-review.d.ts +68 -1
- package/dist/decider-review.d.ts.map +1 -1
- package/dist/decider-review.js +211 -3
- package/dist/decider-review.js.map +1 -1
- package/dist/decision-ledger.d.ts +79 -1
- package/dist/decision-ledger.d.ts.map +1 -1
- package/dist/decision-ledger.js +118 -11
- package/dist/decision-ledger.js.map +1 -1
- package/dist/decision-view.d.ts +41 -0
- package/dist/decision-view.d.ts.map +1 -0
- package/dist/decision-view.js +264 -0
- package/dist/decision-view.js.map +1 -0
- package/dist/exit-codes.d.ts +44 -0
- package/dist/exit-codes.d.ts.map +1 -0
- package/dist/exit-codes.js +49 -0
- package/dist/exit-codes.js.map +1 -0
- package/dist/findings.d.ts +3 -0
- package/dist/findings.d.ts.map +1 -1
- package/dist/findings.js.map +1 -1
- package/dist/index.d.ts +7 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -0
- package/dist/index.js.map +1 -1
- package/dist/manifest.d.ts +2 -2
- package/dist/program-branch.d.ts +65 -0
- package/dist/program-branch.d.ts.map +1 -0
- package/dist/program-branch.js +123 -0
- package/dist/program-branch.js.map +1 -0
- package/dist/programs-dir.d.ts +27 -0
- package/dist/programs-dir.d.ts.map +1 -0
- package/dist/programs-dir.js +52 -0
- package/dist/programs-dir.js.map +1 -0
- package/dist/publish.d.ts +105 -0
- package/dist/publish.d.ts.map +1 -0
- package/dist/publish.js +528 -0
- package/dist/publish.js.map +1 -0
- package/dist/report-path.d.ts +7 -0
- package/dist/report-path.d.ts.map +1 -0
- package/dist/report-path.js +10 -0
- package/dist/report-path.js.map +1 -0
- package/dist/review-pass.d.ts +33 -1
- package/dist/review-pass.d.ts.map +1 -1
- package/dist/review-pass.js +94 -42
- package/dist/review-pass.js.map +1 -1
- package/dist/run-program.d.ts +29 -3
- package/dist/run-program.d.ts.map +1 -1
- package/dist/run-program.js +431 -47
- package/dist/run-program.js.map +1 -1
- package/dist/run-publish.d.ts +41 -0
- package/dist/run-publish.d.ts.map +1 -0
- package/dist/run-publish.js +33 -0
- package/dist/run-publish.js.map +1 -0
- package/package.json +5 -4
- package/skills/plan-program/SKILL.md +14 -0
- package/templates/AGENTS.md +29 -29
- package/templates/CLAUDE.md +7 -7
- package/templates/vision.md +46 -46
package/dist/run-program.js
CHANGED
|
@@ -7,10 +7,16 @@ import { resolveSummary, summaryContract } from "./agent-summary.js";
|
|
|
7
7
|
import { authorProgram } from "./author.js";
|
|
8
8
|
import { decisionContract, decisionFingerprint, extractDecisions, } from "./decision.js";
|
|
9
9
|
import { appendLedgerEvents, readDecisionLedger, } from "./decision-ledger.js";
|
|
10
|
-
import {
|
|
10
|
+
import { escalatedRecords, renderRecord, } from "./decision-view.js";
|
|
11
|
+
import { reviewDecisions, triageFindings } from "./decider-review.js";
|
|
12
|
+
import { fingerprint } from "./findings.js";
|
|
11
13
|
import { findCycles, stableTopologicalOrder } from "./graph.js";
|
|
12
|
-
import { extractFindings, findingsContract, findingsToLedgerEvents, renderPassReport, reviewerAbsentOutcome, runReviewPass, } from "./review-pass.js";
|
|
14
|
+
import { extractFindings, findingsContract, findingsToLedgerEvents, hasRoutableEvidence, locateInRepo, renderPassReport, reviewerAbsentOutcome, runReviewPass, verifyEvidence, } from "./review-pass.js";
|
|
13
15
|
import { loadManifest, saveManifest, } from "./manifest.js";
|
|
16
|
+
import { detectDefaultBranch, programBranchName } from "./program-branch.js";
|
|
17
|
+
import { restoreProgramsDir, snapshotProgramsDir } from "./programs-dir.js";
|
|
18
|
+
import { CouldNotStartError } from "./exit-codes.js";
|
|
19
|
+
import { runReportPath } from "./report-path.js";
|
|
14
20
|
const execFileAsync = promisify(execFile);
|
|
15
21
|
/** Matches every wording git uses to report an empty commit attempt. */
|
|
16
22
|
const NOTHING_TO_COMMIT = /nothing to commit|nothing added to commit|no changes added to commit/u;
|
|
@@ -33,6 +39,15 @@ export const defaultGitOps = {
|
|
|
33
39
|
return undefined;
|
|
34
40
|
}
|
|
35
41
|
},
|
|
42
|
+
async currentBranch(cwd) {
|
|
43
|
+
try {
|
|
44
|
+
const { stdout } = await execFileAsync("git", ["symbolic-ref", "--quiet", "--short", "HEAD"], { cwd });
|
|
45
|
+
return stdout.trim();
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
return undefined;
|
|
49
|
+
}
|
|
50
|
+
},
|
|
36
51
|
async commitAll(cwd, message) {
|
|
37
52
|
await execFileAsync("git", ["add", "-A"], { cwd });
|
|
38
53
|
try {
|
|
@@ -128,6 +143,37 @@ function decisionsRuledOnSection(records) {
|
|
|
128
143
|
lines.push("");
|
|
129
144
|
return lines;
|
|
130
145
|
}
|
|
146
|
+
/**
|
|
147
|
+
* A human-decided finding is binding on the next run, same as a
|
|
148
|
+
* human-decided decision — but the label's force is origin-specific, so the
|
|
149
|
+
* brief renders the *instruction* each label binds, never a bare label
|
|
150
|
+
* (SC-16). `fix-now` is a required correction; `accept` forbids reopening
|
|
151
|
+
* the settled limitation; `escalate` is degenerate for a human ruling — the
|
|
152
|
+
* human is the escalation target — so it binds no implementer instruction,
|
|
153
|
+
* exactly as an escalated-but-unruled finding renders none.
|
|
154
|
+
*/
|
|
155
|
+
function findingsRuledOnSection(records) {
|
|
156
|
+
const lines = [];
|
|
157
|
+
for (const record of records) {
|
|
158
|
+
if (record.status !== "human-decided")
|
|
159
|
+
continue;
|
|
160
|
+
if (record.humanChosen === "fix-now") {
|
|
161
|
+
lines.push(`- **${record.subject}** — binding: the human ruled \`fix-now\` — ` +
|
|
162
|
+
`${record.humanReason}. This is a required correction: address ` +
|
|
163
|
+
"the cited defect in this workstream; it is not optional.");
|
|
164
|
+
}
|
|
165
|
+
else if (record.humanChosen === "accept") {
|
|
166
|
+
lines.push(`- **${record.subject}** — binding: the human ruled \`accept\` — ` +
|
|
167
|
+
`${record.humanReason}. This is a settled known limitation: do ` +
|
|
168
|
+
"not reopen, re-litigate, or attempt to fix it.");
|
|
169
|
+
}
|
|
170
|
+
// A human `escalate` ruling (or any other label) binds no implementer
|
|
171
|
+
// instruction — the human is the escalation target, not a fix to build.
|
|
172
|
+
}
|
|
173
|
+
if (lines.length === 0)
|
|
174
|
+
return [];
|
|
175
|
+
return ["## Findings ruled on", "", ...lines, ""];
|
|
176
|
+
}
|
|
131
177
|
function implementerBrief(manifest, workstream, spec, ledger, priorFailure) {
|
|
132
178
|
const roster = manifest.workstreams
|
|
133
179
|
.map((entry) => {
|
|
@@ -137,6 +183,7 @@ function implementerBrief(manifest, workstream, spec, ledger, priorFailure) {
|
|
|
137
183
|
.join("\n");
|
|
138
184
|
const relevantDecisions = ledger.decisions.filter((record) => record.workstream === workstream.id &&
|
|
139
185
|
(record.status === "human-decided" || record.status === "ratified"));
|
|
186
|
+
const relevantFindings = ledger.findings.filter((record) => record.workstream === workstream.id && record.status === "human-decided");
|
|
140
187
|
return [
|
|
141
188
|
`# Workstream ${workstream.id}: ${workstream.name}`,
|
|
142
189
|
"",
|
|
@@ -149,6 +196,7 @@ function implementerBrief(manifest, workstream, spec, ledger, priorFailure) {
|
|
|
149
196
|
roster,
|
|
150
197
|
"",
|
|
151
198
|
...decisionsRuledOnSection(relevantDecisions),
|
|
199
|
+
...findingsRuledOnSection(relevantFindings),
|
|
152
200
|
...(priorFailure
|
|
153
201
|
? [
|
|
154
202
|
"## Previous attempt failed",
|
|
@@ -216,10 +264,16 @@ export async function runProgram(options) {
|
|
|
216
264
|
const git = options.git ?? defaultGitOps;
|
|
217
265
|
const log = options.log ?? ((line) => console.log(line));
|
|
218
266
|
const now = options.now ?? (() => new Date());
|
|
219
|
-
let manifest
|
|
267
|
+
let manifest;
|
|
268
|
+
try {
|
|
269
|
+
manifest = await loadManifest(root, options.programId);
|
|
270
|
+
}
|
|
271
|
+
catch (error) {
|
|
272
|
+
throw new CouldNotStartError(error.message, { cause: error });
|
|
273
|
+
}
|
|
220
274
|
const resolvedAgent = resolveAgent(config);
|
|
221
275
|
if (!resolvedAgent) {
|
|
222
|
-
throw new
|
|
276
|
+
throw new CouldNotStartError("No implementer configured. Set the `agent` block in nightshift.config.json.");
|
|
223
277
|
}
|
|
224
278
|
const agent = resolvedAgent;
|
|
225
279
|
const recovery = resolveRecoveryAgent(config);
|
|
@@ -236,7 +290,7 @@ export async function runProgram(options) {
|
|
|
236
290
|
if (cycles.length > 0) {
|
|
237
291
|
// A cyclic graph cannot be ordered; this is a planning defect, not a
|
|
238
292
|
// decision the runner can take. Terminate with the diagnosis.
|
|
239
|
-
throw new
|
|
293
|
+
throw new CouldNotStartError(`Dependency cycle(s) in the manifest: ${cycles
|
|
240
294
|
.map((cycle) => cycle.join(" -> "))
|
|
241
295
|
.join("; ")}. Re-plan with /plan-program.`);
|
|
242
296
|
}
|
|
@@ -256,11 +310,49 @@ export async function runProgram(options) {
|
|
|
256
310
|
workstream.status === "in_progress" ||
|
|
257
311
|
workstream.status === "parked");
|
|
258
312
|
if (dirty.length > 0 && !resuming) {
|
|
259
|
-
throw new
|
|
313
|
+
throw new CouldNotStartError(`The working tree has uncommitted changes the run would sweep into its commits:\n` +
|
|
260
314
|
dirty.map((path) => ` ${path}`).join("\n") +
|
|
261
315
|
`\nCommit or stash them, then re-run.`);
|
|
262
316
|
}
|
|
317
|
+
// A run always commits to the program's own branch — never the default
|
|
318
|
+
// branch, and never another program's branch, since the runner commits
|
|
319
|
+
// with `git add -A` and a mismatched worktree would sweep this program's
|
|
320
|
+
// work onto the wrong branch. These refuse before authorProgram below
|
|
321
|
+
// spawns the first agent; --force is the explicit escape hatch.
|
|
322
|
+
if (!options.force) {
|
|
323
|
+
const currentBranch = await git.currentBranch(root);
|
|
324
|
+
const branchCommand = `nightshift branch ${options.programId}`;
|
|
325
|
+
const expectedBranch = programBranchName(options.programId);
|
|
326
|
+
if (currentBranch === undefined) {
|
|
327
|
+
throw new CouldNotStartError(`HEAD is detached; a run must be on this program's branch ` +
|
|
328
|
+
`\`${expectedBranch}\`. Create or switch to it with ` +
|
|
329
|
+
`\`${branchCommand}\`, or re-run with --force.`);
|
|
330
|
+
}
|
|
331
|
+
const defaultBranch = await detectDefaultBranch(root);
|
|
332
|
+
if (currentBranch === defaultBranch) {
|
|
333
|
+
throw new CouldNotStartError(`Refusing to run on the default branch \`${defaultBranch}\`; a ` +
|
|
334
|
+
`program run commits to its own branch. Switch with ` +
|
|
335
|
+
`\`${branchCommand}\`, or re-run with --force.`);
|
|
336
|
+
}
|
|
337
|
+
if (currentBranch !== expectedBranch) {
|
|
338
|
+
throw new CouldNotStartError(`This run is for \`${options.programId}\`, whose branch is ` +
|
|
339
|
+
`\`${expectedBranch}\`, but HEAD is \`${currentBranch}\`. Switch ` +
|
|
340
|
+
`with \`${branchCommand}\`, or re-run with --force.`);
|
|
341
|
+
}
|
|
342
|
+
}
|
|
263
343
|
}
|
|
344
|
+
// Subject ids already sent to the decider this run — shared across the
|
|
345
|
+
// authoring and build stages (a run is one process) and across the
|
|
346
|
+
// decision and finding paths, so "once per subject per run" (SC-10) is
|
|
347
|
+
// honest rather than per-stage. A resumed run starts a fresh set and
|
|
348
|
+
// legitimately re-reviews an earlier run's subjects.
|
|
349
|
+
const reviewed = new Set();
|
|
350
|
+
// Subject ids the decider actually ruled on this run — a strict subset of
|
|
351
|
+
// `reviewed` (a subject can be sent but the invocation fail or return no
|
|
352
|
+
// parseable verdict). This, not `reviewed`, is the run-local basis for the
|
|
353
|
+
// report's "this run" triage ratio — see decider-review.ts's doc comment
|
|
354
|
+
// on `triaged`.
|
|
355
|
+
const triaged = new Set();
|
|
264
356
|
// Authoring runs before building: every workstream whose spec is missing
|
|
265
357
|
// gets one, in dependency order, before anything is implemented. It
|
|
266
358
|
// reloads the manifest afterward because authoring may have merged
|
|
@@ -273,6 +365,8 @@ export async function runProgram(options) {
|
|
|
273
365
|
git,
|
|
274
366
|
log,
|
|
275
367
|
now,
|
|
368
|
+
reviewed,
|
|
369
|
+
triaged,
|
|
276
370
|
});
|
|
277
371
|
manifest = await loadManifest(root, options.programId);
|
|
278
372
|
// Loaded once so every brief in this run projects the same picture of
|
|
@@ -322,13 +416,20 @@ export async function runProgram(options) {
|
|
|
322
416
|
}
|
|
323
417
|
}
|
|
324
418
|
const ledger = await readDecisionLedger(root, options.programId);
|
|
325
|
-
const escalations = ledger
|
|
419
|
+
const escalations = escalatedRecords(ledger);
|
|
326
420
|
const complete = results.every((result) => result.outcome.status === "complete" ||
|
|
327
421
|
result.outcome.status === "skipped");
|
|
328
422
|
manifest.program.status = complete ? "complete" : "partial";
|
|
329
423
|
await saveManifest(root, options.programId, manifest);
|
|
330
|
-
const reportPath =
|
|
331
|
-
await writeFile(reportPath,
|
|
424
|
+
const reportPath = runReportPath(root, options.programId);
|
|
425
|
+
await writeFile(reportPath,
|
|
426
|
+
// `triaged` is the run-local set of subject ids the decider actually
|
|
427
|
+
// ruled on this run (built up across the authoring and build stages,
|
|
428
|
+
// see its declaration above) — exactly the `triagedThisRun` basis the
|
|
429
|
+
// "this run" triage ratio needs, since the projected ledger carries no
|
|
430
|
+
// run identifier of its own. It excludes ids that were merely sent but
|
|
431
|
+
// whose invocation failed or returned no valid verdict.
|
|
432
|
+
renderRunReport(manifest, results, ledger, triaged, authorResult, now()), "utf8");
|
|
332
433
|
if (isRepository) {
|
|
333
434
|
await git.commitPaths(root, `nightshift(${options.programId}): run report and decision ledger`, ["docs/programs"]);
|
|
334
435
|
}
|
|
@@ -413,18 +514,17 @@ export async function runProgram(options) {
|
|
|
413
514
|
// The manifest's single commit field records the workstream's final
|
|
414
515
|
// verified state — after any kept fix, that is the fix commit, not
|
|
415
516
|
// the earlier green one.
|
|
416
|
-
|
|
517
|
+
let finalCommit = critique?.finalCommit ?? c0;
|
|
417
518
|
// Findings anchor to c0 (the green, pre-critique commit) — the
|
|
418
519
|
// honest rollback point — while the decider below diffs from
|
|
419
520
|
// baseCommit (pre-workstream), so it sees the whole workstream.
|
|
420
521
|
const findingEvents = findingsToLedgerEvents({
|
|
421
522
|
workstreamId: workstream.id,
|
|
422
|
-
findings: critique?.outcome.open ?? [],
|
|
523
|
+
findings: (critique?.outcome.open ?? []).filter(hasRoutableEvidence),
|
|
423
524
|
...(c0 === undefined ? {} : { baseCommit: c0 }),
|
|
424
525
|
now,
|
|
425
526
|
});
|
|
426
527
|
await appendLedgerEvents(root, options.programId, findingEvents);
|
|
427
|
-
const findingDecisions = findingEvents.flatMap((event) => event.kind === "decision-recorded" ? [event.decision] : []);
|
|
428
528
|
// The commit's own sha cannot be part of the tree it commits, so the
|
|
429
529
|
// manifest records it only now — swept forward into whatever commits
|
|
430
530
|
// next. Replay reads the manifest's current state, not the commit
|
|
@@ -433,7 +533,24 @@ export async function runProgram(options) {
|
|
|
433
533
|
if (finalCommit !== undefined)
|
|
434
534
|
workstream.commit = finalCommit;
|
|
435
535
|
await saveManifest(root, options.programId, manifest);
|
|
436
|
-
await reviewWorkstreamDecisions(workstream.id,
|
|
536
|
+
await reviewWorkstreamDecisions(workstream.id, parsed.decisions, baseCommit);
|
|
537
|
+
// Findings are triaged against the pre-workstream diff (baseCommit),
|
|
538
|
+
// the same one the decider reviews decisions against — c0/finalCommit
|
|
539
|
+
// are the fix loop's own rollback anchors, a different thing.
|
|
540
|
+
await reviewWorkstreamFindings(workstream.id, findingEvents, baseCommit);
|
|
541
|
+
// A fix-now triage drives exactly one bounded fix attempt, here —
|
|
542
|
+
// after the triage that produced it, before the run advances. Only
|
|
543
|
+
// reachable when there is a green commit to attempt from and a
|
|
544
|
+
// decider that could have produced a fix-now verdict in the first
|
|
545
|
+
// place (SC-13).
|
|
546
|
+
if (finalCommit !== undefined && decider) {
|
|
547
|
+
const driven = await driveFixNowFindings(workstream, spec, critique?.outcome.open ?? [], finalCommit);
|
|
548
|
+
if (driven !== finalCommit) {
|
|
549
|
+
finalCommit = driven;
|
|
550
|
+
workstream.commit = finalCommit;
|
|
551
|
+
await saveManifest(root, options.programId, manifest);
|
|
552
|
+
}
|
|
553
|
+
}
|
|
437
554
|
if (critique)
|
|
438
555
|
base.testCritique = critique.outcome;
|
|
439
556
|
base.testCritiqueDiffClipped = critique?.diffClipped ?? false;
|
|
@@ -484,10 +601,99 @@ export async function runProgram(options) {
|
|
|
484
601
|
agentRunner,
|
|
485
602
|
git,
|
|
486
603
|
isRepository,
|
|
604
|
+
reviewed,
|
|
605
|
+
triaged,
|
|
487
606
|
now,
|
|
488
607
|
log,
|
|
489
608
|
});
|
|
490
609
|
}
|
|
610
|
+
function reviewWorkstreamFindings(workstreamId, findings, baseCommit) {
|
|
611
|
+
return triageFindings({
|
|
612
|
+
root,
|
|
613
|
+
programId: options.programId,
|
|
614
|
+
manifest,
|
|
615
|
+
workstreamId,
|
|
616
|
+
findings,
|
|
617
|
+
baseCommit,
|
|
618
|
+
decider,
|
|
619
|
+
agentRunner,
|
|
620
|
+
git,
|
|
621
|
+
isRepository,
|
|
622
|
+
reviewed,
|
|
623
|
+
triaged,
|
|
624
|
+
now,
|
|
625
|
+
log,
|
|
626
|
+
});
|
|
627
|
+
}
|
|
628
|
+
/**
|
|
629
|
+
* Drives the one bounded fix attempt a `fix-now` triage earns (WS-06).
|
|
630
|
+
* Selects findings from the ledger projection — never from any in-memory
|
|
631
|
+
* disposition — so a human-decided finding (status `"human-decided"`, not
|
|
632
|
+
* `"fix-now"`) is never handed to the fix seam (SC-14). A workstream's
|
|
633
|
+
* `fix-now` findings are fixed together, in one `attemptFix` call, then
|
|
634
|
+
* recorded as one `finding-fix-attempted` event per finding id — a ledger
|
|
635
|
+
* entry, never a re-triage (SC-14). Returns the (possibly unchanged) green
|
|
636
|
+
* commit; never throws, never blocks the run (SC-15).
|
|
637
|
+
*/
|
|
638
|
+
async function driveFixNowFindings(workstream, spec, openFindings, greenCommit) {
|
|
639
|
+
const ledger = await readDecisionLedger(root, options.programId);
|
|
640
|
+
const fixNowIds = new Set(ledger.findings
|
|
641
|
+
.filter((record) => record.workstream === workstream.id && record.status === "fix-now")
|
|
642
|
+
.map((record) => record.id));
|
|
643
|
+
if (fixNowIds.size === 0)
|
|
644
|
+
return greenCommit;
|
|
645
|
+
const toFix = [];
|
|
646
|
+
const matchedIds = [];
|
|
647
|
+
for (const candidate of openFindings) {
|
|
648
|
+
const id = fingerprint({ ...candidate, workstreamId: workstream.id });
|
|
649
|
+
if (!fixNowIds.has(id))
|
|
650
|
+
continue;
|
|
651
|
+
toFix.push(candidate);
|
|
652
|
+
matchedIds.push(id);
|
|
653
|
+
}
|
|
654
|
+
// An id triaged fix-now but absent from the open set (should not
|
|
655
|
+
// happen — it is where the id came from) is skipped, fail-open.
|
|
656
|
+
if (toFix.length === 0)
|
|
657
|
+
return greenCommit;
|
|
658
|
+
const fix = await attemptFix({
|
|
659
|
+
root,
|
|
660
|
+
programId: options.programId,
|
|
661
|
+
workstream,
|
|
662
|
+
spec,
|
|
663
|
+
config,
|
|
664
|
+
agentRunner,
|
|
665
|
+
verifyRunner,
|
|
666
|
+
git,
|
|
667
|
+
agent,
|
|
668
|
+
findings: toFix,
|
|
669
|
+
greenCommit,
|
|
670
|
+
label: "fix-now fix",
|
|
671
|
+
log,
|
|
672
|
+
});
|
|
673
|
+
// "kept" only when the attempt both verified clean and actually landed a
|
|
674
|
+
// commit — a clean-but-empty attempt is a decline, not a fix, and must
|
|
675
|
+
// reach the human exactly like a failed verification does.
|
|
676
|
+
const kept = fix.outcome === "kept" && fix.commit !== undefined;
|
|
677
|
+
const note = kept
|
|
678
|
+
? fix.summary
|
|
679
|
+
: fix.outcome === "kept"
|
|
680
|
+
? `${fix.summary} (the attempt verified clean but made no change; no fix landed)`
|
|
681
|
+
: `${fix.summary} (fix failed verification and was discarded: ${fix.failure})`;
|
|
682
|
+
const events = matchedIds.map((id) => ({
|
|
683
|
+
kind: "finding-fix-attempted",
|
|
684
|
+
at: now().toISOString(),
|
|
685
|
+
id,
|
|
686
|
+
outcome: kept ? "kept" : "failed",
|
|
687
|
+
note,
|
|
688
|
+
...(kept && fix.commit !== undefined ? { commit: fix.commit } : {}),
|
|
689
|
+
attemptedBy: "implementer",
|
|
690
|
+
}));
|
|
691
|
+
await appendLedgerEvents(root, options.programId, events);
|
|
692
|
+
log(kept
|
|
693
|
+
? `${workstream.id}: fix-now fix verified and committed`
|
|
694
|
+
: `${workstream.id}: fix-now fix failed and was escalated — ${note}`);
|
|
695
|
+
return kept ? fix.greenCommit : greenCommit;
|
|
696
|
+
}
|
|
491
697
|
}
|
|
492
698
|
/** Undefined means the attempt verified clean; otherwise the diagnosis. */
|
|
493
699
|
async function verifyAttempt(config, verifyRunner, root, agentExitCode) {
|
|
@@ -675,6 +881,55 @@ function testCritiqueFixBrief(workstream, spec, findings) {
|
|
|
675
881
|
summaryContract(),
|
|
676
882
|
].join("\n");
|
|
677
883
|
}
|
|
884
|
+
/**
|
|
885
|
+
* The fix seam, in full: one brief carrying every finding, one implementer
|
|
886
|
+
* invocation, one verification, one commit on green or one reset on
|
|
887
|
+
* failure. Never invokes a reviewer and never loops — the caller owns
|
|
888
|
+
* whether (and how many times) this is called.
|
|
889
|
+
*/
|
|
890
|
+
async function attemptFix(options) {
|
|
891
|
+
const { root, programId, workstream, spec, config, agentRunner, verifyRunner, git, agent, findings, greenCommit, label, } = options;
|
|
892
|
+
const brief = testCritiqueFixBrief(workstream, spec, findings);
|
|
893
|
+
const invocation = await invokeAgent(agentRunner, agent, brief, root);
|
|
894
|
+
const summary = resolveSummary(invocation.output).text;
|
|
895
|
+
const failure = await verifyAttempt(config, verifyRunner, root, invocation.exitCode);
|
|
896
|
+
if (failure === undefined) {
|
|
897
|
+
// The runner's own pending ledger and manifest writes under
|
|
898
|
+
// docs/programs sit uncommitted in the tree until the run's final
|
|
899
|
+
// commit — appended by earlier steps in this very workstream (findings,
|
|
900
|
+
// triage) or by an earlier workstream in the same run. A plain
|
|
901
|
+
// `git add -A` would otherwise sweep that bookkeeping into this fix's
|
|
902
|
+
// commit and make a genuine no-op look like a landed fix. Only a change
|
|
903
|
+
// outside docs/programs counts as the implementer having done anything;
|
|
904
|
+
// with none, skip committing — the pending bookkeeping stays for a
|
|
905
|
+
// later commit to pick up, nothing is lost.
|
|
906
|
+
const dirty = await git.dirtyPaths(root);
|
|
907
|
+
const meaningfulChange = dirty.some((path) => !path.replaceAll("\\", "/").startsWith("docs/programs/"));
|
|
908
|
+
const commit = meaningfulChange
|
|
909
|
+
? await git.commitAll(root, `nightshift(${programId}): ${workstream.id} ${label}`)
|
|
910
|
+
: undefined;
|
|
911
|
+
return {
|
|
912
|
+
outcome: "kept",
|
|
913
|
+
greenCommit: commit ?? greenCommit,
|
|
914
|
+
...(commit === undefined ? {} : { commit }),
|
|
915
|
+
summary,
|
|
916
|
+
};
|
|
917
|
+
}
|
|
918
|
+
// The same uncommitted bookkeeping the clean path guards against sweeping
|
|
919
|
+
// into a commit is, on this path, what a whole-tree `git reset --hard`
|
|
920
|
+
// would silently throw away: by the time a fix-now attempt runs, this
|
|
921
|
+
// workstream's finding-recorded, finding-triaged and decision-reviewed
|
|
922
|
+
// events are on disk and not yet committed. Resetting without them would
|
|
923
|
+
// erase the finding this very attempt is about to report a failure on —
|
|
924
|
+
// and the `finding-fix-attempted` event appended afterwards would project
|
|
925
|
+
// against nothing, so the escalation would never reach the human (SC-15).
|
|
926
|
+
// Preserve the journal across the rollback exactly as `decide`'s replay
|
|
927
|
+
// does; the run's final commit picks it up.
|
|
928
|
+
const snapshot = await snapshotProgramsDir(root);
|
|
929
|
+
await git.resetHard(root, greenCommit);
|
|
930
|
+
await restoreProgramsDir(root, snapshot);
|
|
931
|
+
return { outcome: "failed", greenCommit, summary, failure };
|
|
932
|
+
}
|
|
678
933
|
/**
|
|
679
934
|
* The test-critique driver: wires WS-01's bounded review loop to the two
|
|
680
935
|
* briefs above, and owns the green-state invariant (SC-07) — a closure
|
|
@@ -689,6 +944,7 @@ async function runTestCritique(options) {
|
|
|
689
944
|
if (!reviewer) {
|
|
690
945
|
return { outcome: reviewerAbsentOutcome(), finalCommit: greenCommit, diffClipped };
|
|
691
946
|
}
|
|
947
|
+
const locate = (file) => locateInRepo(root, file);
|
|
692
948
|
const review = async (_round, priorOpen) => {
|
|
693
949
|
const rawDiff = baseCommit !== undefined ? await git.diffSince(root, baseCommit) : "";
|
|
694
950
|
const diff = clipForReview(rawDiff, "diff");
|
|
@@ -698,24 +954,33 @@ async function runTestCritique(options) {
|
|
|
698
954
|
const brief = testCritiqueReviewerBrief(manifest, workstream, clippedSpec.text, diff.text, priorOpen);
|
|
699
955
|
const invocation = await invokeAgent(agentRunner, reviewer, brief, root);
|
|
700
956
|
const parsed = extractFindings(invocation.output);
|
|
957
|
+
const findings = verifyEvidence(parsed.findings, locate);
|
|
701
958
|
const ran = invocation.exitCode === 0 && hasFindingsBlock(invocation.output);
|
|
702
|
-
return { findings
|
|
959
|
+
return { findings, errors: parsed.errors, ran };
|
|
703
960
|
};
|
|
704
961
|
const respond = async (_round, findings) => {
|
|
705
|
-
const
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
|
|
962
|
+
const fix = await attemptFix({
|
|
963
|
+
root,
|
|
964
|
+
programId: manifest.program.id,
|
|
965
|
+
workstream,
|
|
966
|
+
spec,
|
|
967
|
+
config,
|
|
968
|
+
agentRunner,
|
|
969
|
+
verifyRunner,
|
|
970
|
+
git,
|
|
971
|
+
agent,
|
|
972
|
+
findings,
|
|
973
|
+
greenCommit,
|
|
974
|
+
label: "test critique fix",
|
|
975
|
+
log,
|
|
976
|
+
});
|
|
977
|
+
greenCommit = fix.greenCommit;
|
|
978
|
+
if (fix.outcome === "kept") {
|
|
713
979
|
log(`${workstream.id}: test critique fix verified and committed`);
|
|
714
|
-
return { note: `${summary} (fix verified and committed)` };
|
|
980
|
+
return { note: `${fix.summary} (fix verified and committed)` };
|
|
715
981
|
}
|
|
716
|
-
|
|
717
|
-
|
|
718
|
-
return { note: `${summary} (fix failed verification and was discarded; the green state was preserved)` };
|
|
982
|
+
log(`${workstream.id}: test critique fix failed verification and was discarded — ${fix.failure}`);
|
|
983
|
+
return { note: `${fix.summary} (fix failed verification and was discarded; the green state was preserved)` };
|
|
719
984
|
};
|
|
720
985
|
const outcome = await runReviewPass({ review, respond });
|
|
721
986
|
return { outcome, finalCommit: greenCommit, diffClipped };
|
|
@@ -751,7 +1016,128 @@ function renderSpecsSection(authorResult) {
|
|
|
751
1016
|
}
|
|
752
1017
|
return lines;
|
|
753
1018
|
}
|
|
754
|
-
|
|
1019
|
+
/**
|
|
1020
|
+
* By subject + message + first evidence location — the dedup key for
|
|
1021
|
+
* collecting minor/advisory findings into known-limitations. Deliberately
|
|
1022
|
+
* looser than `findings.ts`'s `fingerprint` (which is identity across
|
|
1023
|
+
* review rounds): here two textually distinct findings about the same
|
|
1024
|
+
* subject and location are still worth listing once.
|
|
1025
|
+
*/
|
|
1026
|
+
function limitationKey(finding) {
|
|
1027
|
+
const first = finding.evidence[0];
|
|
1028
|
+
const location = first === undefined
|
|
1029
|
+
? ""
|
|
1030
|
+
: first.kind === "location"
|
|
1031
|
+
? `${first.file}:${first.startLine}`
|
|
1032
|
+
: first.kind === "concern"
|
|
1033
|
+
? first.named
|
|
1034
|
+
: `${first.metric}:${first.value}`;
|
|
1035
|
+
return `${finding.subject}|${finding.message}|${location}`;
|
|
1036
|
+
}
|
|
1037
|
+
function renderRawFinding(finding, workstreamLabel, sinceFixed) {
|
|
1038
|
+
const suffix = sinceFixed ? " (raised, since fixed)" : "";
|
|
1039
|
+
return `- **${finding.subject}** (${finding.severity}, ${workstreamLabel}) — ${finding.message}${suffix}`;
|
|
1040
|
+
}
|
|
1041
|
+
/**
|
|
1042
|
+
* A pass's minor/advisory findings, `open` and `resolved` alike (SC-07's
|
|
1043
|
+
* "every ... raised" — a finding the writer fixed was still raised),
|
|
1044
|
+
* deduped within the pass's own union so a finding present in both (should
|
|
1045
|
+
* WS-02's own dedup ever let that happen) is not double-listed.
|
|
1046
|
+
*/
|
|
1047
|
+
function collectPassLimitations(outcome, workstreamLabel) {
|
|
1048
|
+
if (!outcome)
|
|
1049
|
+
return [];
|
|
1050
|
+
const seen = new Set();
|
|
1051
|
+
const lines = [];
|
|
1052
|
+
for (const finding of outcome.open) {
|
|
1053
|
+
if (finding.severity !== "minor" && finding.severity !== "advisory")
|
|
1054
|
+
continue;
|
|
1055
|
+
const key = limitationKey(finding);
|
|
1056
|
+
if (seen.has(key))
|
|
1057
|
+
continue;
|
|
1058
|
+
seen.add(key);
|
|
1059
|
+
lines.push(renderRawFinding(finding, workstreamLabel, false));
|
|
1060
|
+
}
|
|
1061
|
+
for (const finding of outcome.resolved) {
|
|
1062
|
+
if (finding.severity !== "minor" && finding.severity !== "advisory")
|
|
1063
|
+
continue;
|
|
1064
|
+
const key = limitationKey(finding);
|
|
1065
|
+
if (seen.has(key))
|
|
1066
|
+
continue;
|
|
1067
|
+
seen.add(key);
|
|
1068
|
+
lines.push(renderRawFinding(finding, workstreamLabel, true));
|
|
1069
|
+
}
|
|
1070
|
+
return lines;
|
|
1071
|
+
}
|
|
1072
|
+
/**
|
|
1073
|
+
* One aggregated section (SC-07): every minor/advisory finding raised
|
|
1074
|
+
* anywhere in the run, every finding the decider accepted, any finding
|
|
1075
|
+
* still awaiting triage or already ruled on by a human, the outcome of
|
|
1076
|
+
* every fix-now attempt, and the run-local triage ratio. Fail-open
|
|
1077
|
+
* throughout — an empty run renders a plain statement, never nothing.
|
|
1078
|
+
*/
|
|
1079
|
+
function renderKnownLimitations(results, authorResult, ledger, escalated, triagedThisRun, programId) {
|
|
1080
|
+
const lines = ["## Known limitations", ""];
|
|
1081
|
+
const passLimitations = [
|
|
1082
|
+
...results.flatMap((result) => collectPassLimitations(result.testCritique, `${result.id} test critique`)),
|
|
1083
|
+
...authorResult.results.flatMap((entry) => collectPassLimitations(entry.specCritique, `${entry.id} spec critique`)),
|
|
1084
|
+
];
|
|
1085
|
+
const acceptedFindings = ledger.findings.filter((record) => record.status === "accepted");
|
|
1086
|
+
const openFindings = ledger.findings.filter((record) => record.status === "open");
|
|
1087
|
+
const humanDecidedFindings = ledger.findings.filter((record) => record.status === "human-decided");
|
|
1088
|
+
const fixNowFindings = ledger.findings.filter((record) => record.status === "fix-now");
|
|
1089
|
+
const hasFindingsContent = passLimitations.length > 0 ||
|
|
1090
|
+
acceptedFindings.length > 0 ||
|
|
1091
|
+
openFindings.length > 0 ||
|
|
1092
|
+
humanDecidedFindings.length > 0;
|
|
1093
|
+
if (!hasFindingsContent) {
|
|
1094
|
+
lines.push("No known limitations were recorded.", "");
|
|
1095
|
+
}
|
|
1096
|
+
else {
|
|
1097
|
+
if (passLimitations.length > 0) {
|
|
1098
|
+
lines.push("Minor and advisory findings raised during review:", "", ...passLimitations, "");
|
|
1099
|
+
}
|
|
1100
|
+
if (acceptedFindings.length > 0) {
|
|
1101
|
+
lines.push("Accepted as known limitations:", "");
|
|
1102
|
+
for (const record of acceptedFindings) {
|
|
1103
|
+
lines.push(...renderRecord(record, { density: "compact", programId }), "");
|
|
1104
|
+
}
|
|
1105
|
+
}
|
|
1106
|
+
if (openFindings.length > 0) {
|
|
1107
|
+
lines.push("Recorded, not yet triaged:", "");
|
|
1108
|
+
for (const record of openFindings) {
|
|
1109
|
+
lines.push(...renderRecord(record, { density: "compact", programId }), "");
|
|
1110
|
+
}
|
|
1111
|
+
}
|
|
1112
|
+
if (humanDecidedFindings.length > 0) {
|
|
1113
|
+
lines.push("Ruled by the human:", "");
|
|
1114
|
+
for (const record of humanDecidedFindings) {
|
|
1115
|
+
lines.push(...renderRecord(record, { density: "compact", programId }), "");
|
|
1116
|
+
}
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
if (fixNowFindings.length > 0) {
|
|
1120
|
+
lines.push("Fix-now attempts:", "");
|
|
1121
|
+
for (const record of fixNowFindings) {
|
|
1122
|
+
if (record.fixAttempt?.outcome === "kept") {
|
|
1123
|
+
const commitPart = record.fixAttempt.commit
|
|
1124
|
+
? ` (commit \`${record.fixAttempt.commit}\`)`
|
|
1125
|
+
: "";
|
|
1126
|
+
lines.push(`- **${record.subject}** — fixed in the run${commitPart}: ${record.fixAttempt.note}`, "");
|
|
1127
|
+
}
|
|
1128
|
+
else {
|
|
1129
|
+
lines.push(`- **${record.subject}** — fix-now triaged; outcome pending.`, "");
|
|
1130
|
+
}
|
|
1131
|
+
}
|
|
1132
|
+
}
|
|
1133
|
+
const triagedCount = triagedThisRun.size;
|
|
1134
|
+
const escalatedThisRunCount = escalated.filter((record) => triagedThisRun.has(record.id)).length;
|
|
1135
|
+
lines.push(`The decider triaged ${triagedCount} subject${triagedCount === 1 ? "" : "s"} this run; ` +
|
|
1136
|
+
`${escalatedThisRunCount} ${escalatedThisRunCount === 1 ? "was" : "were"} escalated.`, "");
|
|
1137
|
+
return lines;
|
|
1138
|
+
}
|
|
1139
|
+
export function renderRunReport(manifest, results, ledger, triagedThisRun, authorResult, at) {
|
|
1140
|
+
const programId = manifest.program.id;
|
|
755
1141
|
const built = results.filter((result) => result.outcome.status === "complete" ||
|
|
756
1142
|
result.outcome.status === "skipped").length;
|
|
757
1143
|
const lines = [
|
|
@@ -763,12 +1149,16 @@ function renderRunReport(manifest, results, decisions, escalations, authorResult
|
|
|
763
1149
|
"",
|
|
764
1150
|
...renderSpecsSection(authorResult),
|
|
765
1151
|
];
|
|
766
|
-
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
1152
|
+
const escalated = escalatedRecords(ledger);
|
|
1153
|
+
if (escalated.length > 0) {
|
|
1154
|
+
lines.push("## Needs your attention", "", "The decider reviewed these choices and findings and believes you", "might decide differently. Each is self-contained below: why it is", "here, the alternatives, and a command to flip it if you disagree.", "");
|
|
1155
|
+
for (const record of escalated) {
|
|
1156
|
+
lines.push(...renderRecord(record, { density: "full", programId }));
|
|
1157
|
+
if ("origin" in record &&
|
|
1158
|
+
record.origin === "finding" &&
|
|
1159
|
+
record.fixAttempt?.outcome === "failed") {
|
|
1160
|
+
lines.push(`- **Fix attempt:** ${record.fixAttempt.note}`, "");
|
|
1161
|
+
}
|
|
772
1162
|
}
|
|
773
1163
|
}
|
|
774
1164
|
const failures = results.filter((result) => result.outcome.status === "failed" || result.outcome.status === "parked");
|
|
@@ -797,23 +1187,17 @@ function renderRunReport(manifest, results, decisions, escalations, authorResult
|
|
|
797
1187
|
}
|
|
798
1188
|
}
|
|
799
1189
|
lines.push("");
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
.filter((option) => option.label !== record.decision.chosen)
|
|
805
|
-
.map((option) => option.label)
|
|
806
|
-
.join(", ");
|
|
807
|
-
lines.push(`### ${record.decision.title} (${record.id}) — ${record.status}`, "", `- **Workstream:** ${record.workstream}${record.oneWay ? " · **one-way door**" : ""}`, `- **Context:** ${record.decision.context}`, `- **Chosen:** ${record.decision.chosen} — ${record.decision.rationale}`, `- **Alternatives:** ${alternatives === "" ? "(none listed)" : alternatives}`, ...(record.reviewRationale
|
|
808
|
-
? [`- **Decider:** ${record.reviewRationale}`]
|
|
809
|
-
: []), ...(record.baseCommit
|
|
810
|
-
? [`- **Anchor commit:** \`${record.baseCommit}\``]
|
|
811
|
-
: []), "");
|
|
812
|
-
}
|
|
1190
|
+
const settledDecisions = ledger.decisions.filter((record) => record.status !== "escalated");
|
|
1191
|
+
lines.push("## Decisions made along the way", "", "Every judgment call an agent surfaced, with its review status.", "Anything here can be revisited: roll back to the anchor commit and", "re-run, or just say which option you want changed.", "");
|
|
1192
|
+
if (settledDecisions.length === 0) {
|
|
1193
|
+
lines.push("None surfaced.", "");
|
|
813
1194
|
}
|
|
814
1195
|
else {
|
|
815
|
-
|
|
1196
|
+
for (const record of settledDecisions) {
|
|
1197
|
+
lines.push(...renderRecord(record, { density: "compact", programId }), "");
|
|
1198
|
+
}
|
|
816
1199
|
}
|
|
1200
|
+
lines.push(...renderKnownLimitations(results, authorResult, ledger, escalated, triagedThisRun, programId));
|
|
817
1201
|
const decisionErrors = results.flatMap((result) => result.decisionErrors);
|
|
818
1202
|
if (decisionErrors.length > 0) {
|
|
819
1203
|
lines.push("## Decision blocks the runner could not read", "", ...decisionErrors.map((error) => `- ${error}`), "");
|