pi-plans 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +5 -12
  3. package/README.md +5 -5
  4. package/agents/execution-reviewer.md +92 -0
  5. package/index.ts +13 -23
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +14 -7
  8. package/references/plan-artifact-template.md +11 -1
  9. package/references/state-and-config.md +3 -3
  10. package/scripts/validate.ts +20 -3
  11. package/src/auditor.ts +306 -63
  12. package/src/code-graph/commands.ts +6 -1
  13. package/src/dashboard.ts +91 -13
  14. package/src/exec.ts +835 -142
  15. package/src/plan.ts +1 -1
  16. package/src/refine-ui-state.ts +1 -1
  17. package/src/refine-ui.ts +40 -8
  18. package/src/resume-command.ts +19 -3
  19. package/src/resume.ts +5 -1
  20. package/src/staleness.ts +53 -0
  21. package/src/state.ts +1 -0
  22. package/src/task-tool.ts +1 -1
  23. package/src/tasks.ts +62 -5
  24. package/src/ui-language.ts +4 -0
  25. package/src/workflow-state.ts +93 -6
  26. package/tests/analyze-refs.test.ts +1 -1
  27. package/tests/auditor.test.ts +299 -16
  28. package/tests/dashboard.test.ts +202 -2
  29. package/tests/exec-review-loop.test.ts +724 -0
  30. package/tests/exec.test.ts +198 -44
  31. package/tests/extension-load.test.ts +1 -1
  32. package/tests/refine-ui.test.ts +25 -2
  33. package/tests/resume-lifecycle.test.ts +5 -1
  34. package/tests/resume.test.ts +6 -0
  35. package/tests/staleness.test.ts +76 -0
  36. package/tests/state.test.ts +4 -0
  37. package/tests/tasks.test.ts +142 -0
  38. package/tests/workflow-state.test.ts +105 -0
  39. package/tools/analyze-refs.ts +17 -6
  40. package/tools/execute-plan.ts +12 -5
  41. package/tools/plans.ts +1 -1
  42. package/tools/refine.ts +22 -3
@@ -0,0 +1,724 @@
1
+ /**
2
+ * Execution-review loop tests (v0.8): detached rounds, abort/identity
3
+ * lifecycle, the fingerprint guard, budget accounting (commit-only), round
4
+ * report persistence, the cap pause, and resume self-heal.
5
+ */
6
+
7
+ import * as assert from "node:assert/strict";
8
+ import * as fs from "node:fs";
9
+ import * as os from "node:os";
10
+ import * as path from "node:path";
11
+ import { after, before, describe, it } from "node:test";
12
+ import {
13
+ __awaitReviewRoundForTests,
14
+ __setAuditRunnerForTests,
15
+ getExecution,
16
+ persistTaskProgress,
17
+ restoreFromSession,
18
+ startExecution,
19
+ stopExecution,
20
+ } from "../src/exec.ts";
21
+ import { setMessagingApi } from "../src/messaging.ts";
22
+ import { applyTaskUpdate } from "../src/task-tool.ts";
23
+ import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
24
+ import { getRun, initState, startRun } from "../src/state.ts";
25
+ import { createCheckpoint, loadCheckpoint, mutateCheckpoint, applyExecutionApproved, applyExecutionProgress, applyPlanWritten, planIdentityOf } from "../src/workflow-state.ts";
26
+
27
+ const PLAN = `# PLAN_v1 - review-loop fixture
28
+
29
+ ## Tasks
30
+
31
+ - Task-1: engine — files: lib/engine.js; wave: 1
32
+ - Task-2: report — files: lib/report.js; wave: 1
33
+
34
+ ## Verification Checks
35
+
36
+ - [ ] \`VC-001\` covers \`Task-1\`; pass condition: engine works; evidence: tests; metric: green.
37
+ - [ ] \`VC-002\` covers \`Task-2\`; pass condition: report written; evidence: file; metric: green.
38
+ `;
39
+
40
+ let root = "";
41
+ let counter = 0;
42
+
43
+ before(() => {
44
+ root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-review-loop-"));
45
+ });
46
+
47
+ after(() => {
48
+ fs.rmSync(root, { recursive: true, force: true });
49
+ __setAuditRunnerForTests(null);
50
+ });
51
+
52
+ function freshWorkdir(): { workdir: string; planPath: string; runId: string } {
53
+ counter += 1;
54
+ const workdir = path.join(root, `repo-${counter}`);
55
+ fs.mkdirSync(workdir, { recursive: true });
56
+ fs.mkdirSync(path.join(workdir, "lib"), { recursive: true });
57
+ // Real covered files so the round fingerprint tracks their mtimes.
58
+ fs.writeFileSync(path.join(workdir, "lib", "engine.js"), "export const engine = 1;\n", "utf8");
59
+ fs.writeFileSync(path.join(workdir, "lib", "report.js"), "export const report = 1;\n", "utf8");
60
+ initState(workdir);
61
+ const { run } = startRun(workdir, { topic: `r${counter}`, skill: "plan-small", requestText: "demo" });
62
+ createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
63
+ const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
64
+ fs.mkdirSync(run.artifact_dir, { recursive: true });
65
+ fs.writeFileSync(planPath, PLAN, "utf8");
66
+ mutateCheckpoint(workdir, run.run_id, (cp) =>
67
+ applyExecutionApproved(
68
+ applyPlanWritten({ ...cp, nextAction: "accept-execute" }, planIdentityOf(planPath, 1)),
69
+ { plan: planIdentityOf(planPath, 1), worktree: workdir, headAtApproval: null, approvedAt: cp.updatedAt },
70
+ ),
71
+ );
72
+ return { workdir, planPath, runId: run.run_id };
73
+ }
74
+
75
+ function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }, components?: RefineOverlayComponent[]) {
76
+ const entries: Array<{ customType: string; data?: unknown; content?: string }> = [];
77
+ const ctx = {
78
+ cwd: workdir,
79
+ sessionManager: {},
80
+ hasUI: true,
81
+ mode,
82
+ entries,
83
+ ui: {
84
+ notify: () => {},
85
+ setStatus: () => {},
86
+ setWidget: () => {},
87
+ theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
88
+ // Minimal overlay host: counts ui.custom opens (one per controller)
89
+ // and captures the rendered component so tests can drive its input.
90
+ custom: (render: (tui: unknown, theme: unknown, kb: unknown, done: () => void) => { handleInput(data: string): void }) => {
91
+ if (customOpens) customOpens.count += 1;
92
+ const component = render({ requestRender() {}, terminal: undefined }, { fg: (_c: string, t: string) => t, bold: (t: string) => t }, undefined, () => {});
93
+ if (components) components.push(component);
94
+ return Promise.resolve();
95
+ },
96
+ },
97
+ isIdle: () => true,
98
+ hasPendingMessages: () => false,
99
+ } as never;
100
+ setMessagingApi({
101
+ appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }),
102
+ sendMessage: (message: { customType: string; content: string }) => entries.push({ customType: message.customType, content: message.content }),
103
+ } as never);
104
+ return ctx;
105
+ }
106
+
107
+ async function startTerminal(planPath: string, workdir: string) {
108
+ const ctx = makeCtx(workdir);
109
+ await startExecution(ctx, { planPath, planTasks: (await import("../src/plan.ts")).parsePlanTasks(PLAN), items: (await import("../src/plan.ts")).parseChecklist(PLAN) });
110
+ for (const id of ["Task-1", "Task-2"]) {
111
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
112
+ persistTaskProgress(ctx);
113
+ }
114
+ return ctx;
115
+ }
116
+
117
+ /** A runner whose rounds the test resolves by hand. */
118
+ function controlledRunner() {
119
+ const pending: Array<(value: { round: number; passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null) => void> = [];
120
+ let calls = 0;
121
+ const runner = async () => {
122
+ calls += 1;
123
+ return new Promise<{ round: number; passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null>((resolve) => pending.push(resolve));
124
+ };
125
+ const resolveRound = (value: Parameters<typeof pending[0]>[0]) => {
126
+ const resolve = pending.shift();
127
+ if (resolve) resolve(value);
128
+ };
129
+ /** Resolve every still-pending round (teardown: the identity guard discards them). */
130
+ const drainAll = (value: Parameters<typeof pending[0]>[0] = null) => {
131
+ while (pending.length > 0) pending.shift()!(value);
132
+ };
133
+ return { runner, resolveRound, drainAll, calls: () => calls };
134
+ }
135
+
136
+ /** Let the engine chain advance one step without awaiting its completion. */
137
+ const tick = async (): Promise<void> => {
138
+ await new Promise<void>((resolve) => setImmediate(resolve));
139
+ await new Promise<void>((resolve) => setImmediate(resolve));
140
+ };
141
+
142
+ describe("execution-review loop (v0.8)", () => {
143
+ it("tui/rpc detach: the settle returns before the round resolves, status becomes verifying", async () => {
144
+ const { workdir, planPath, runId } = freshWorkdir();
145
+ await startTerminal(planPath, workdir);
146
+ const ctl = controlledRunner();
147
+ __setAuditRunnerForTests(ctl.runner);
148
+ const ctxTui = makeCtx(workdir, "tui");
149
+ await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
150
+ // The restore returned while the round is STILL running: in-flight marker
151
+ // present, run status moved to verifying, no outcome committed yet.
152
+ assert.ok(getExecution()!.review.inFlight, "the round is in flight after the settle returned");
153
+ assert.equal(getExecution()!.audit.rounds, 0, "no budget spent yet");
154
+ assert.equal(getRun(workdir, runId)?.status, "verifying", "run status enters verifying");
155
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
156
+ await __awaitReviewRoundForTests();
157
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed", "the round completes the run after resolution");
158
+ __setAuditRunnerForTests(null);
159
+ });
160
+
161
+ it("print/json keep the inline await: the settle returns only after the round commits", async () => {
162
+ const { workdir, planPath, runId } = freshWorkdir();
163
+ await startTerminal(planPath, workdir);
164
+ const ctl = controlledRunner();
165
+ __setAuditRunnerForTests(ctl.runner);
166
+ const ctxPrint = makeCtx(workdir, "print");
167
+ const restoring = restoreFromSession(ctxPrint, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
168
+ await tick(); // the inline round spawns
169
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
170
+ await restoring;
171
+ // Inline: by the time restoreFromSession returned, the round committed —
172
+ // and a passing round completes (clearing execution state).
173
+ assert.equal(getExecution(), null, "the inline round committed before the settle returned");
174
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
175
+ assert.equal(ctl.calls(), 1);
176
+ __setAuditRunnerForTests(null);
177
+ });
178
+
179
+ it("a cancelled round burns no budget, sends no wake, and pauses nothing", async () => {
180
+ const { workdir, planPath } = freshWorkdir();
181
+ const ctx = await startTerminal(planPath, workdir);
182
+ __setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
183
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
184
+ const ex = getExecution()!;
185
+ assert.equal(ex.audit.rounds, 0, "cancelled rounds burn no budget");
186
+ assert.equal(ex.stall.paused, false, "cancelled rounds do not pause");
187
+ assert.equal(ctx.entries.filter((e) => String(e.customType).startsWith("pi-plans-audit-")).length, 0, "no wake, no report message");
188
+ await stopExecution(ctx, "teardown");
189
+ });
190
+
191
+ it("fingerprint change discards the attempt, re-runs without budget burn, and both report files survive", async () => {
192
+ const { workdir, planPath, runId } = freshWorkdir();
193
+ const ctx = await startTerminal(planPath, workdir);
194
+ const ctl = controlledRunner();
195
+ __setAuditRunnerForTests(ctl.runner);
196
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
197
+ await tick(); // the inline round spawns
198
+ // Mutate a covered file while the round is in flight (real mutation path).
199
+ const engine = path.join(workdir, "lib", "engine.js");
200
+ fs.writeFileSync(engine, "export const engine = 2;\n", "utf8");
201
+ const later = new Date(Date.now() + 10_000);
202
+ fs.utimesSync(engine, later, later);
203
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "stale verdict" });
204
+ await tick();
205
+ // Attempt 1 discarded (report on disk, marked), attempt 2 committed the pass.
206
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "fresh verdict" });
207
+ await restoring;
208
+ await __awaitReviewRoundForTests();
209
+ // Attempt 1 discarded (report on disk, marked), attempt 2 committed the pass.
210
+ const ex = getExecution();
211
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
212
+ const runDir = path.join(workdir, ".git", "pi-plans", "runs", runId);
213
+ const attempt1 = path.join(runDir, "execution-review", "round-1-attempt-1.md");
214
+ const attempt2 = path.join(runDir, "execution-review", "round-1-attempt-2.md");
215
+ assert.ok(fs.existsSync(attempt1), "the discarded attempt's report survives");
216
+ assert.match(fs.readFileSync(attempt1, "utf8"), /outcome: discarded/);
217
+ assert.match(fs.readFileSync(attempt1, "utf8"), /discarded: fingerprint changed/);
218
+ assert.ok(fs.existsSync(attempt2), "the re-run's report exists beside it");
219
+ assert.match(fs.readFileSync(attempt2, "utf8"), /outcome: passed/);
220
+ assert.equal(ctl.calls(), 2, "one discard plus one re-run");
221
+ assert.ok(!ex, "execution completed");
222
+ __setAuditRunnerForTests(null);
223
+ });
224
+
225
+ it("two consecutive discards commit as a budget-counting undeterminable round", async () => {
226
+ const { workdir, planPath } = freshWorkdir();
227
+ const ctx = await startTerminal(planPath, workdir);
228
+ const ctl = controlledRunner();
229
+ __setAuditRunnerForTests(ctl.runner);
230
+ const engine = path.join(workdir, "lib", "engine.js");
231
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
232
+ await tick();
233
+ const discard = () => {
234
+ fs.writeFileSync(engine, `export const engine = ${Math.random()};\n`, "utf8");
235
+ const later = new Date(Date.now() + 60_000);
236
+ fs.utimesSync(engine, later, later);
237
+ };
238
+ discard();
239
+ ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale 1" });
240
+ await tick();
241
+ assert.equal(getExecution()!.audit.rounds, 0, "the first discard burns nothing");
242
+ discard();
243
+ ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale 2" });
244
+ await tick();
245
+ assert.equal(getExecution()!.audit.rounds, 1, "the second consecutive discard commits as undeterminable (budget spent)");
246
+ assert.deepEqual(getExecution()!.audit.undeterminable, ["VC-001", "VC-002"]);
247
+ await stopExecution(ctx, "teardown");
248
+ ctl.drainAll();
249
+ await restoring;
250
+ await __awaitReviewRoundForTests();
251
+ __setAuditRunnerForTests(null);
252
+ });
253
+
254
+ it("a failed round rolls back, returns the run to executing, and wakes exactly once", async () => {
255
+ const { workdir, planPath, runId } = freshWorkdir();
256
+ const ctx = await startTerminal(planPath, workdir);
257
+ const ctl = controlledRunner();
258
+ __setAuditRunnerForTests(ctl.runner);
259
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
260
+ await tick();
261
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "VC-002 is not satisfied" });
262
+ await restoring;
263
+ await __awaitReviewRoundForTests();
264
+ const ex = getExecution()!;
265
+ assert.equal(getRun(workdir, runId)?.status, "executing", "rollback returns the run to executing for repair");
266
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending", "the failed check's task rolled back");
267
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
268
+ assert.equal(wakes.length, 1, "exactly one wake per committed failed outcome");
269
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing", "checkpoint phase stays executing across the round");
270
+ await stopExecution(ctx, "teardown");
271
+ __setAuditRunnerForTests(null);
272
+ });
273
+
274
+ it("a second restore aborts the previous in-flight round; never two live rounds on one run", async () => {
275
+ const { workdir, planPath } = freshWorkdir();
276
+ await startTerminal(planPath, workdir);
277
+ const ctl = controlledRunner();
278
+ __setAuditRunnerForTests(ctl.runner);
279
+ const ctx = makeCtx(workdir, "tui");
280
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
281
+ const firstAttempt = getExecution()!.review.attempts;
282
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
283
+ const after = getExecution()!;
284
+ assert.ok(after.review.inFlight, "the second resume started its own round");
285
+ assert.equal(after.review.attempts, 1, "the replacement identity starts its own attempt 1 (the old round was aborted)");
286
+ assert.notEqual(after.review.inFlight.controller, undefined, "exactly one live round — its abort lifecycle is owned");
287
+ // The aborted first round resolves late against a replaced identity: it
288
+ // must be discarded silently (no extra wake, no budget charge).
289
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "late stale round" });
290
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "fresh round" });
291
+ await __awaitReviewRoundForTests();
292
+ assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-complete").length, 1, "the run completed exactly once");
293
+ __setAuditRunnerForTests(null);
294
+ });
295
+
296
+ it("every attempt opens a fresh overlay; the reopen shortcut is inert with no in-flight round", async () => {
297
+ const { workdir, planPath } = freshWorkdir();
298
+ await startTerminal(planPath, workdir);
299
+ const ctl = controlledRunner();
300
+ __setAuditRunnerForTests(ctl.runner);
301
+ const opens = { count: 0 };
302
+ const components: Array<{ handleInput(data: string): void }> = [];
303
+ const ctx = makeCtx(workdir, "tui", opens, components);
304
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
305
+ assert.equal(opens.count, 1, "the round opened its overlay before spawning");
306
+ // ESC closed the (one-shot) controller — only THEN may reopen rebuild it
307
+ // (the anti-stacking guard keeps a second overlay off a live one).
308
+ components[0]!.handleInput("\x1b");
309
+ const { reopenReviewOverlay } = await import("../src/exec.ts");
310
+ reopenReviewOverlay(ctx);
311
+ assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round after ESC");
312
+ // A second reopen while the new controller is live must NOT stack.
313
+ reopenReviewOverlay(ctx);
314
+ assert.equal(opens.count, 2, "reopen never stacks a second live overlay");
315
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "done" });
316
+ await __awaitReviewRoundForTests();
317
+ // No round in flight → the shortcut is inert.
318
+ reopenReviewOverlay(ctx);
319
+ assert.equal(opens.count, 2, "reopen is inert with no in-flight round");
320
+ __setAuditRunnerForTests(null);
321
+ });
322
+
323
+ it("a legacy cap pause (old prefix) survives restore paused at its round count", async () => {
324
+ const { workdir, planPath, runId } = freshWorkdir();
325
+ await startTerminal(planPath, workdir);
326
+ // Simulate a checkpoint paused by a v0.7 build ("completion audit
327
+ // exhausted 3 rounds") — the dual-matched prefix must keep it paused.
328
+ mutateCheckpoint(workdir, runId, (cp) =>
329
+ applyExecutionProgress(cp, { audit: { rounds: 3, lastResult: "VC-001" }, pausedReason: "completion audit exhausted 3 rounds (failed: VC-001)." }),
330
+ );
331
+ const load = await import("../src/exec.ts");
332
+ const result = load.loadExecutionFromCheckpoint(makeCtx(workdir), runId);
333
+ assert.equal(result.status, "loaded", `checkpoint loads (${result.status})`);
334
+ const ex = getExecution()!;
335
+ assert.equal(ex.stall.paused, true, "a legacy cap pause stays paused across restore");
336
+ assert.equal(ex.audit.rounds, 3, "the legacy round count survives — restore grants no budget");
337
+ await stopExecution(makeCtx(workdir), "teardown");
338
+ });
339
+ });
340
+
341
+ describe("findings-driven fix loop (v0.9)", () => {
342
+ const finding = (id: string, severity: "high" | "medium" | "low", taskIds: string[], extra: Partial<{ note: string; proposedTask: string }> = {}) => ({
343
+ id, severity, taskIds, note: extra.note ?? `${id} note`, evidence: "src/lib", raw: `- \` ${id}\` raw`,
344
+ ...(extra.proposedTask ? { proposedTask: extra.proposedTask } : {}),
345
+ });
346
+
347
+ it("all VCs pass but a mapped high finding blocks completion: rollback + exactly one wake + NOT done", async () => {
348
+ const { workdir, planPath, runId } = freshWorkdir();
349
+ const ctx = await startTerminal(planPath, workdir);
350
+ const ctl = controlledRunner();
351
+ __setAuditRunnerForTests(ctl.runner);
352
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
353
+ await tick();
354
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "ok but findings", findings: [finding("F-001", "high", ["Task-2"])] } as never);
355
+ await restoring;
356
+ await __awaitReviewRoundForTests();
357
+ const ex = getExecution()!;
358
+ assert.ok(ex, "high findings never complete the run (liveness)");
359
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
360
+ assert.equal(getRun(workdir, runId)?.status, "executing", "back to executing for the fix round");
361
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending", "the high finding's mapped task rolled back");
362
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
363
+ assert.equal(wakes.length, 1, "exactly one wake");
364
+ assert.match(String(wakes[0].content), /1 high-severity finding/);
365
+ assert.match(String(wakes[0].content), /F-001/);
366
+ assert.match(String(wakes[0].content), /Full round report: /, "the wake references the round report path");
367
+ await stopExecution(ctx, "teardown");
368
+ ctl.drainAll();
369
+ });
370
+
371
+ it("keep-done asymmetry: a pure-high rollback keeps earlier VC passes; a VC-fail rollback invalidates", async () => {
372
+ const { workdir, planPath } = freshWorkdir();
373
+ const ctx = await startTerminal(planPath, workdir);
374
+ const ctl = controlledRunner();
375
+ __setAuditRunnerForTests(ctl.runner);
376
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
377
+ await tick();
378
+ // Round 1: VC-001 passes, VC-002 passes, but F-001 (high) maps to Task-1 —
379
+ // which VC-001 covers. The high rollback must NOT clear VC-001's done.
380
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001", "high", ["Task-1"])] } as never);
381
+ await restoring;
382
+ await __awaitReviewRoundForTests();
383
+ const ex = getExecution()!;
384
+ assert.equal(ex.items.find((i) => i.id === "VC-001")?.done, true, "pure-high rollback keeps the earlier pass");
385
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
386
+ await stopExecution(ctx, "teardown");
387
+ ctl.drainAll();
388
+
389
+ // Contrast: a VC-fail rollback invalidates checks covering the reopened task.
390
+ const second = freshWorkdir();
391
+ const ctx2 = await startTerminal(second.planPath, second.workdir);
392
+ const ctl2 = controlledRunner();
393
+ __setAuditRunnerForTests(ctl2.runner);
394
+ const restoring2 = restoreFromSession(ctx2, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
395
+ await tick();
396
+ ctl2.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "r1" });
397
+ await restoring2;
398
+ await __awaitReviewRoundForTests();
399
+ const ex2 = getExecution()!;
400
+ assert.equal(ex2.items.find((i) => i.id === "VC-001")?.done, true, "unrelated pass kept");
401
+ assert.equal(ex2.tasks.find((t) => t.id === "Task-2")?.status, "pending", "failed check's task rolled back");
402
+ await stopExecution(ctx2, "teardown");
403
+ ctl2.drainAll();
404
+ });
405
+
406
+ it("undeterminable round carrying a high finding still wakes (high wins over self-schedule)", async () => {
407
+ const { workdir, planPath } = freshWorkdir();
408
+ const ctx = await startTerminal(planPath, workdir);
409
+ const ctl = controlledRunner();
410
+ __setAuditRunnerForTests(ctl.runner);
411
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
412
+ await tick();
413
+ ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "unreadable", findings: [finding("F-001", "high", ["Task-1"])] } as never);
414
+ await restoring;
415
+ await __awaitReviewRoundForTests();
416
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
417
+ assert.equal(wakes.length, 1, "the high branch wakes despite all-undeterminable verdicts");
418
+ const ex = getExecution()!;
419
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
420
+ assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "undeterminable set still recorded");
421
+ await stopExecution(ctx, "teardown");
422
+ ctl.drainAll();
423
+ });
424
+
425
+ it("an unmapped high appends a plan task (proposed-task applied mechanically) and wakes once", async () => {
426
+ const { workdir, planPath, runId } = freshWorkdir();
427
+ const ctx = await startTerminal(planPath, workdir);
428
+ const ctl = controlledRunner();
429
+ __setAuditRunnerForTests(ctl.runner);
430
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
431
+ await tick();
432
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-009", "high", [], { proposedTask: "harden the retry budget guard" })] } as never);
433
+ await restoring;
434
+ await __awaitReviewRoundForTests();
435
+ const ex = getExecution()!;
436
+ const amended = ex.tasks.find((t) => t.id === "Task-3");
437
+ assert.ok(amended, "the unmapped high gained an appended task");
438
+ assert.equal(amended.status, "pending");
439
+ assert.match(amended.title, /fix F-009: harden the retry budget guard/);
440
+ assert.match(amended.title, /appended by execution review round 1/);
441
+ const planText = fs.readFileSync(planPath, "utf8");
442
+ assert.match(planText, /- `Task-3`: fix F-009: harden the retry budget guard/, "the plan file carries the appended bullet");
443
+ assert.match(planText, /## Verification Checks/, "the plan stays parseable (section intact)");
444
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
445
+ assert.ok(cp.execution?.tasks?.["Task-3"], "checkpoint carries the appended task");
446
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
447
+ assert.equal(wakes.length, 1);
448
+ assert.match(String(wakes[0].content), /Tasks appended to the plan for unmapped findings: Task-3/);
449
+ await stopExecution(ctx, "teardown");
450
+ ctl.drainAll();
451
+ });
452
+
453
+ it("completion with residual medium/low findings summarizes them in the completion message", async () => {
454
+ const { workdir, planPath, runId } = freshWorkdir();
455
+ const ctx = await startTerminal(planPath, workdir);
456
+ const ctl = controlledRunner();
457
+ __setAuditRunnerForTests(ctl.runner);
458
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
459
+ await tick();
460
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "clean", findings: [finding("F-002", "medium", []), finding("F-003", "low", ["Task-1"])] } as never);
461
+ await restoring;
462
+ await __awaitReviewRoundForTests();
463
+ assert.equal(getExecution(), null, "no high findings: the run completes");
464
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
465
+ const done = ctx.entries.filter((e) => e.customType === "pi-plans-complete");
466
+ assert.equal(done.length, 1);
467
+ assert.match(String(done[0].content), /Recorded findings that did not block completion: F-002 \(medium\), F-003 \(low\)/);
468
+ ctl.drainAll();
469
+ });
470
+
471
+ it("findings persist across the session snapshot and survive the fresh-budget renewal", async () => {
472
+ const { workdir, planPath } = freshWorkdir();
473
+ const ctx = await startTerminal(planPath, workdir);
474
+ const ctl = controlledRunner();
475
+ __setAuditRunnerForTests(ctl.runner);
476
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
477
+ await tick();
478
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "r1", findings: [finding("F-001", "high", ["Task-2"])] } as never);
479
+ await restoring;
480
+ await __awaitReviewRoundForTests();
481
+ const ex = getExecution()!;
482
+ assert.equal(ex.audit.findings.length, 1, "findings in live state");
483
+ // The snapshot round-trip: a session restore rebuilds them.
484
+ const restored = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
485
+ await tick();
486
+ await restored;
487
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings survive the session restore");
488
+ // Renewal: /plans-execute grants a fresh budget and keeps the findings.
489
+ const { resumeActiveExecution } = await import("../src/exec.ts");
490
+ // Drive rounds 2-5 through the real path: fix, re-close, settle (restore
491
+ // is the settle entry in these tests) — the same high persists each time.
492
+ for (let r = 2; r <= 5; r++) {
493
+ for (const id of ["Task-1", "Task-2"]) {
494
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `fix round ${r}`);
495
+ }
496
+ persistTaskProgress(ctx);
497
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
498
+ await tick();
499
+ ctl.resolveRound({ round: r, passed: [], failed: [], undeterminable: [], report: `r${r}`, findings: [finding("F-001", "high", ["Task-2"])] } as never);
500
+ await settle;
501
+ await __awaitReviewRoundForTests();
502
+ }
503
+ // Round 5 committed with the high: the next terminal cycle hits the cap.
504
+ for (const id of ["Task-1", "Task-2"]) {
505
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", "post-cap close");
506
+ }
507
+ persistTaskProgress(ctx);
508
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
509
+ const paused = getExecution()!;
510
+ assert.equal(paused.stall.paused, true, "the cap pause fired");
511
+ assert.match(String(paused.stall.pausedReason), /high findings: F-001/, "the pause reason names the high findings");
512
+ assert.match(String(paused.stall.pausedReason), /execution review exhausted 5 rounds/, "the pause prefix phrase survives");
513
+ // Cancelled rounds burn no budget and loop nowhere — safe to leave the
514
+ // runner in this mode while checking the renewal semantics.
515
+ __setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
516
+ assert.ok(resumeActiveExecution(ctx), "renewal lifts the pause");
517
+ assert.equal(getExecution()!.audit.rounds, 0, "fresh budget");
518
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "stable ids carry into the fresh budget");
519
+ ctl.drainAll();
520
+ await stopExecution(ctx, "teardown");
521
+ });
522
+ });
523
+
524
+ describe("mixed and hygiene rounds (v0.9.1 F-004/F-006/F-007/F-012)", () => {
525
+ it("a round with a failed check AND a high finding rolls back the union once and wakes exactly once (F-007)", async () => {
526
+ const { workdir, planPath, runId } = freshWorkdir();
527
+ const ctx = await startTerminal(planPath, workdir);
528
+ const ctl = controlledRunner();
529
+ __setAuditRunnerForTests(ctl.runner);
530
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
531
+ await tick();
532
+ // VC-001 fails (covers Task-1) and F-001 also maps to Task-1: the union
533
+ // must dedupe to one reopen of Task-1 plus Task-2 (VC-002 stays passed).
534
+ ctl.resolveRound({ round: 1, passed: ["VC-002"], failed: ["VC-001"], undeterminable: [], report: "mixed", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-1"], note: "n", evidence: "e", raw: "r" }] } as never);
535
+ await restoring;
536
+ await __awaitReviewRoundForTests();
537
+ const ex = getExecution()!;
538
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "Task-1 reopened once by both channels");
539
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "complete", "the passing check's task stays closed");
540
+ assert.equal(ex.items.find((i) => i.id === "VC-002")?.done, true, "unrelated pass kept");
541
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
542
+ assert.equal(wakes.length, 1, "one wake for the mixed round");
543
+ assert.match(String(wakes[0].content), /and failed checks: VC-001/);
544
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
545
+ await stopExecution(ctx, "teardown");
546
+ ctl.drainAll();
547
+ });
548
+
549
+ it("a pure VC-fail round keeps the v0.8 lead — never '0 high-severity findings' (F-004)", async () => {
550
+ const { workdir, planPath } = freshWorkdir();
551
+ const ctx = await startTerminal(planPath, workdir);
552
+ const ctl = controlledRunner();
553
+ __setAuditRunnerForTests(ctl.runner);
554
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
555
+ await tick();
556
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "vc2 broken" });
557
+ await restoring;
558
+ await __awaitReviewRoundForTests();
559
+ const wake = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
560
+ assert.equal(wake.length, 1);
561
+ assert.doesNotMatch(String(wake[0].content), /0 high-severity finding/);
562
+ assert.doesNotMatch(String(wake[0].content), /High findings:\n\(none/);
563
+ assert.match(String(wake[0].content), /round 1 failed\*\* — checks: VC-002/);
564
+ await stopExecution(ctx, "teardown");
565
+ ctl.drainAll();
566
+ });
567
+
568
+ it("the appended bullet carries its wave tail and sanitizes reviewer text (F-006/F-012)", async () => {
569
+ const { workdir, planPath } = freshWorkdir();
570
+ const ctx = await startTerminal(planPath, workdir);
571
+ const ctl = controlledRunner();
572
+ __setAuditRunnerForTests(ctl.runner);
573
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
574
+ await tick();
575
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden — the retry; budget guard", note: "n", evidence: "e", raw: "r" }] } as never);
576
+ await restoring;
577
+ await __awaitReviewRoundForTests();
578
+ const ex = getExecution()!;
579
+ const appended = ex.tasks.find((t) => t.id === "Task-3");
580
+ assert.ok(appended, "task appended");
581
+ const planText = fs.readFileSync(planPath, "utf8");
582
+ const bullet = planText.split("\n").find((l) => l.startsWith("- `Task-3`:"))!;
583
+ assert.match(bullet, /— wave: \d+$/, "the bullet carries the wave tail");
584
+ // Re-parse restores the same wave the live tree assigned (not wave 1).
585
+ const reparse = (await import("../src/plan.ts")).parsePlanTasks(planText);
586
+ const flat = reparse.tasks.flatMap(function walk(t: { children: unknown[] }) { return [t, ...t.children]; } as never) as never[];
587
+ const reparsed = flat.find((t: { id: string }) => t.id === "Task-3") as { wave: number; title: string; files: string[] };
588
+ assert.equal(reparsed.wave, appended.wave, "re-parse restores the live wave");
589
+ // Sanitized: em dash -> hyphen, ';' -> ',', no forged fields.
590
+ assert.ok(!/—|—/.test(reparsed.title.split("(appended")[0]), "em dashes sanitized out of the reviewer text");
591
+ assert.equal(reparsed.files.length, 0, "no fields forged from reviewer text");
592
+ await stopExecution(ctx, "teardown");
593
+ ctl.drainAll();
594
+ });
595
+ });
596
+
597
+ describe("no-report rounds preserve findings (v0.9.1 F-001)", () => {
598
+ const finding = (id: string) => ({ id, severity: "high" as const, taskIds: ["Task-2"], note: `${id} note`, evidence: "e", raw: "r" });
599
+
600
+ it("a spawn-failure round never vacuously completes a run with an unresolved high", async () => {
601
+ const { workdir, planPath } = freshWorkdir();
602
+ const ctx = await startTerminal(planPath, workdir);
603
+ const ctl = controlledRunner();
604
+ __setAuditRunnerForTests(ctl.runner);
605
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
606
+ await tick();
607
+ // Round 1: both VCs pass, one mapped high -> rollback + wake.
608
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
609
+ await restoring;
610
+ await __awaitReviewRoundForTests();
611
+ const wakesAfterR1 = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length;
612
+ // The executor fixes and re-closes; the settle starts round 2.
613
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
614
+ persistTaskProgress(ctx);
615
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
616
+ await tick();
617
+ // Round 2's subagent FAILS (null outcome): findings must be preserved,
618
+ // no completion, no second wake — the loop self-schedules. The inline
619
+ // settle chain stays pending through the self-scheduled round 3, so
620
+ // resolve round 3 BEFORE awaiting the settle.
621
+ ctl.resolveRound(null);
622
+ await tick();
623
+ const ex = getExecution()!;
624
+ assert.ok(ex, "a spawn-failure round never completes the run");
625
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "unresolved findings survive the no-report round");
626
+ assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length, wakesAfterR1, "no extra wake on the no-report round");
627
+ // Round 3 spawned (self-schedule) and reports the fix — now it completes.
628
+ ctl.resolveRound({ round: 3, passed: [], failed: [], undeterminable: [], report: "fixed", findings: [] } as never);
629
+ await settle;
630
+ await __awaitReviewRoundForTests();
631
+ assert.equal(getExecution(), null, "a clean re-report completes");
632
+ ctl.drainAll();
633
+ });
634
+
635
+ it("the two-consecutive-discard synthesis preserves findings instead of clearing them", async () => {
636
+ const { workdir, planPath } = freshWorkdir();
637
+ const ctx = await startTerminal(planPath, workdir);
638
+ const ctl = controlledRunner();
639
+ __setAuditRunnerForTests(ctl.runner);
640
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
641
+ await tick();
642
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
643
+ await restoring;
644
+ await __awaitReviewRoundForTests();
645
+ const engine = path.join(workdir, "lib", "engine.js");
646
+ const discard = () => {
647
+ fs.writeFileSync(engine, `export const engine = ${Math.random()};\n`, "utf8");
648
+ const later = new Date(Date.now() + 60_000);
649
+ fs.utimesSync(engine, later, later);
650
+ };
651
+ // Re-close, settle, then two fingerprint discards -> the synthesized
652
+ // commit must carry the previous findings forward, not wipe them.
653
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
654
+ persistTaskProgress(ctx);
655
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
656
+ await tick();
657
+ discard();
658
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
659
+ await tick();
660
+ discard();
661
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
662
+ await tick();
663
+ await tick();
664
+ const ex = getExecution()!;
665
+ assert.ok(ex, "the synthesis never completes the run");
666
+ assert.equal(ex.audit.rounds, 2, "the synthesis committed as round 2");
667
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings preserved through the discard synthesis");
668
+ await stopExecution(ctx, "teardown");
669
+ ctl.drainAll();
670
+ await settle;
671
+ await __awaitReviewRoundForTests();
672
+ });
673
+ });
674
+
675
+ describe("plan amendment re-stamps the checkpoint identity (v0.9.1 F-002)", () => {
676
+ it("an amended plan passes /resume-plans instead of plan-mismatch", async () => {
677
+ const { workdir, planPath, runId } = freshWorkdir();
678
+ const ctx = await startTerminal(planPath, workdir);
679
+ const ctl = controlledRunner();
680
+ __setAuditRunnerForTests(ctl.runner);
681
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
682
+ await tick();
683
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden the guard", note: "n", evidence: "e", raw: "r" }] } as never);
684
+ await restoring;
685
+ await __awaitReviewRoundForTests();
686
+ const ex = getExecution()!;
687
+ assert.ok(ex.tasks.find((t) => t.id === "Task-3"), "the amendment appended Task-3");
688
+ // The checkpoint identity now matches the AMENDED file, with provenance.
689
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
690
+ const { sha256File } = await import("../src/workflow-state.ts");
691
+ assert.equal(cp.plan?.sha256, sha256File(planPath), "identity re-stamped to the amended digest");
692
+ assert.equal(cp.execution?.planAmended?.round, 1);
693
+ assert.equal(cp.execution?.planAmended?.sha256, sha256File(planPath));
694
+ // A later resume accepts the amended plan (no plan-mismatch re-approval).
695
+ await stopExecution(ctx, "teardown");
696
+ ctl.drainAll();
697
+ const { loadExecutionFromCheckpoint } = await import("../src/exec.ts");
698
+ const result = loadExecutionFromCheckpoint(makeCtx(workdir), runId);
699
+ assert.equal(result.status, "loaded", `resume accepts the amended plan (${result.status})`);
700
+ assert.deepEqual(result.findings?.map((f) => f.id), ["F-009"], "the load result surfaces unresolved findings for the resume brief (F-005)");
701
+ await stopExecution(makeCtx(workdir), "post-check teardown");
702
+ });
703
+ });
704
+
705
+ describe("executor injection with findings (v0.9)", () => {
706
+ it("executionContextMessage lists unresolved high findings for the repairing agent", async () => {
707
+ const { workdir, planPath } = freshWorkdir();
708
+ const ctx = await startTerminal(planPath, workdir);
709
+ const ctl = controlledRunner();
710
+ __setAuditRunnerForTests(ctl.runner);
711
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
712
+ await tick();
713
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "loop misses union", evidence: "e", raw: "r" }] } as never);
714
+ await restoring;
715
+ await __awaitReviewRoundForTests();
716
+ const { executionContextMessage } = await import("../src/exec.ts");
717
+ const msg = executionContextMessage(ctx) ?? "";
718
+ assert.match(msg, /unresolved high-severity findings/);
719
+ assert.match(msg, /- F-001 \(Task-2\): loop misses union/);
720
+ assert.match(msg, /Fix them, then re-close the affected tasks/);
721
+ await stopExecution(ctx, "teardown");
722
+ ctl.drainAll();
723
+ });
724
+ });