pi-plans 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,17 +12,24 @@ import { after, before, describe, it } from "node:test";
12
12
  import {
13
13
  __awaitReviewRoundForTests,
14
14
  __setAuditRunnerForTests,
15
+ EXECUTION_BLOCKED_CUSTOM_TYPE,
16
+ EXECUTION_CONTINUE_CUSTOM_TYPE,
17
+ executionContextMessage,
15
18
  getExecution,
16
19
  persistTaskProgress,
20
+ registerExecutionTurnHandlers,
17
21
  restoreFromSession,
18
22
  startExecution,
19
23
  stopExecution,
20
24
  } from "../src/exec.ts";
25
+ import { parseChecklist, parsePlanTasks } from "../src/plan.ts";
21
26
  import { setMessagingApi } from "../src/messaging.ts";
22
27
  import { applyTaskUpdate } from "../src/task-tool.ts";
23
- import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
24
- import { getRun, initState, startRun } from "../src/state.ts";
25
- import { createCheckpoint, loadCheckpoint, mutateCheckpoint, applyExecutionApproved, applyExecutionProgress, applyPlanWritten, planIdentityOf } from "../src/workflow-state.ts";
28
+ import { spawnSync } from "node:child_process";
29
+ import { getRun, initState, setRunStatus, startRun } from "../src/state.ts";
30
+ import { createCheckpoint, loadCheckpoint, mutateCheckpoint, applyExecutionApproved, applyExecutionProgress, applyPlanWritten, planIdentityOf, resolveHeadAt } from "../src/workflow-state.ts";
31
+
32
+ import { loadExecutionFromCheckpoint } from "../src/exec.ts";
26
33
 
27
34
  const PLAN = `# PLAN_v1 - review-loop fixture
28
35
 
@@ -72,7 +79,7 @@ function freshWorkdir(): { workdir: string; planPath: string; runId: string } {
72
79
  return { workdir, planPath, runId: run.run_id };
73
80
  }
74
81
 
75
- function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }) {
82
+ function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }, components?: RefineOverlayComponent[]) {
76
83
  const entries: Array<{ customType: string; data?: unknown; content?: string }> = [];
77
84
  const ctx = {
78
85
  cwd: workdir,
@@ -85,9 +92,12 @@ function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?:
85
92
  setStatus: () => {},
86
93
  setWidget: () => {},
87
94
  theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
88
- // Minimal overlay host: counts ui.custom opens (one per controller).
89
- custom: () => {
95
+ // Minimal overlay host: counts ui.custom opens (one per controller)
96
+ // and captures the rendered component so tests can drive its input.
97
+ custom: (render: (tui: unknown, theme: unknown, kb: unknown, done: () => void) => { handleInput(data: string): void }) => {
90
98
  if (customOpens) customOpens.count += 1;
99
+ const component = render({ requestRender() {}, terminal: undefined }, { fg: (_c: string, t: string) => t, bold: (t: string) => t }, undefined, () => {});
100
+ if (components) components.push(component);
91
101
  return Promise.resolve();
92
102
  },
93
103
  },
@@ -144,6 +154,10 @@ describe("execution-review loop (v0.8)", () => {
144
154
  __setAuditRunnerForTests(ctl.runner);
145
155
  const ctxTui = makeCtx(workdir, "tui");
146
156
  await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
157
+ // v0.9.3: a TUI host gets the budget panel before round 1; this minimal
158
+ // host resolves it without a choice, so one hop passes before the round
159
+ // spawns. The detach contract itself is unchanged.
160
+ await tick();
147
161
  // The restore returned while the round is STILL running: in-flight marker
148
162
  // present, run status moved to verifying, no outcome committed yet.
149
163
  assert.ok(getExecution()!.review.inFlight, "the round is in flight after the settle returned");
@@ -275,7 +289,12 @@ describe("execution-review loop (v0.8)", () => {
275
289
  __setAuditRunnerForTests(ctl.runner);
276
290
  const ctx = makeCtx(workdir, "tui");
277
291
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
292
+ // v0.9.3: the first restore resolves the budget (one hop) before its
293
+ // round spawns; wait for it so the second restore really replaces a
294
+ // LIVE round.
295
+ await tick();
278
296
  const firstAttempt = getExecution()!.review.attempts;
297
+ assert.equal(firstAttempt, 1, "the first round is in flight before the second restore");
279
298
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
280
299
  const after = getExecution()!;
281
300
  assert.ok(after.review.inFlight, "the second resume started its own round");
@@ -296,14 +315,19 @@ describe("execution-review loop (v0.8)", () => {
296
315
  const ctl = controlledRunner();
297
316
  __setAuditRunnerForTests(ctl.runner);
298
317
  const opens = { count: 0 };
299
- const ctx = makeCtx(workdir, "tui", opens);
318
+ const components: Array<{ handleInput(data: string): void }> = [];
319
+ const ctx = makeCtx(workdir, "tui", opens, components);
300
320
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
301
321
  assert.equal(opens.count, 1, "the round opened its overlay before spawning");
302
- // ESC closed the (one-shot) controller; the reopen shortcut rebuilds it
303
- // from engine-held lane state while the round is still in flight.
322
+ // ESC closed the (one-shot) controller — only THEN may reopen rebuild it
323
+ // (the anti-stacking guard keeps a second overlay off a live one).
324
+ components[0]!.handleInput("\x1b");
304
325
  const { reopenReviewOverlay } = await import("../src/exec.ts");
305
326
  reopenReviewOverlay(ctx);
306
- assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round");
327
+ assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round after ESC");
328
+ // A second reopen while the new controller is live must NOT stack.
329
+ reopenReviewOverlay(ctx);
330
+ assert.equal(opens.count, 2, "reopen never stacks a second live overlay");
307
331
  ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "done" });
308
332
  await __awaitReviewRoundForTests();
309
333
  // No round in flight → the shortcut is inert.
@@ -329,3 +353,1000 @@ describe("execution-review loop (v0.8)", () => {
329
353
  await stopExecution(makeCtx(workdir), "teardown");
330
354
  });
331
355
  });
356
+
357
+ describe("findings-driven fix loop (v0.9)", () => {
358
+ const finding = (id: string, severity: "high" | "medium" | "low", taskIds: string[], extra: Partial<{ note: string; proposedTask: string }> = {}) => ({
359
+ id, severity, taskIds, note: extra.note ?? `${id} note`, evidence: "src/lib", raw: `- \` ${id}\` raw`,
360
+ ...(extra.proposedTask ? { proposedTask: extra.proposedTask } : {}),
361
+ });
362
+
363
+ it("all VCs pass but a mapped high finding blocks completion: rollback + exactly one wake + NOT done", async () => {
364
+ const { workdir, planPath, runId } = freshWorkdir();
365
+ const ctx = await startTerminal(planPath, workdir);
366
+ const ctl = controlledRunner();
367
+ __setAuditRunnerForTests(ctl.runner);
368
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
369
+ await tick();
370
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "ok but findings", findings: [finding("F-001", "high", ["Task-2"])] } as never);
371
+ await restoring;
372
+ await __awaitReviewRoundForTests();
373
+ const ex = getExecution()!;
374
+ assert.ok(ex, "high findings never complete the run (liveness)");
375
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
376
+ assert.equal(getRun(workdir, runId)?.status, "executing", "back to executing for the fix round");
377
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending", "the high finding's mapped task rolled back");
378
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
379
+ assert.equal(wakes.length, 1, "exactly one wake");
380
+ assert.match(String(wakes[0].content), /1 high-severity finding/);
381
+ assert.match(String(wakes[0].content), /F-001/);
382
+ assert.match(String(wakes[0].content), /Full round report: /, "the wake references the round report path");
383
+ await stopExecution(ctx, "teardown");
384
+ ctl.drainAll();
385
+ });
386
+
387
+ it("keep-done asymmetry: a pure-high rollback keeps earlier VC passes; a VC-fail rollback invalidates", async () => {
388
+ const { workdir, planPath } = freshWorkdir();
389
+ const ctx = await startTerminal(planPath, workdir);
390
+ const ctl = controlledRunner();
391
+ __setAuditRunnerForTests(ctl.runner);
392
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
393
+ await tick();
394
+ // Round 1: VC-001 passes, VC-002 passes, but F-001 (high) maps to Task-1 —
395
+ // which VC-001 covers. The high rollback must NOT clear VC-001's done.
396
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001", "high", ["Task-1"])] } as never);
397
+ await restoring;
398
+ await __awaitReviewRoundForTests();
399
+ const ex = getExecution()!;
400
+ assert.equal(ex.items.find((i) => i.id === "VC-001")?.done, true, "pure-high rollback keeps the earlier pass");
401
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
402
+ await stopExecution(ctx, "teardown");
403
+ ctl.drainAll();
404
+
405
+ // Contrast: a VC-fail rollback invalidates checks covering the reopened task.
406
+ const second = freshWorkdir();
407
+ const ctx2 = await startTerminal(second.planPath, second.workdir);
408
+ const ctl2 = controlledRunner();
409
+ __setAuditRunnerForTests(ctl2.runner);
410
+ const restoring2 = restoreFromSession(ctx2, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
411
+ await tick();
412
+ ctl2.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "r1" });
413
+ await restoring2;
414
+ await __awaitReviewRoundForTests();
415
+ const ex2 = getExecution()!;
416
+ assert.equal(ex2.items.find((i) => i.id === "VC-001")?.done, true, "unrelated pass kept");
417
+ assert.equal(ex2.tasks.find((t) => t.id === "Task-2")?.status, "pending", "failed check's task rolled back");
418
+ await stopExecution(ctx2, "teardown");
419
+ ctl2.drainAll();
420
+ });
421
+
422
+ it("undeterminable round carrying a high finding still wakes (high wins over self-schedule)", async () => {
423
+ const { workdir, planPath } = freshWorkdir();
424
+ const ctx = await startTerminal(planPath, workdir);
425
+ const ctl = controlledRunner();
426
+ __setAuditRunnerForTests(ctl.runner);
427
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
428
+ await tick();
429
+ ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "unreadable", findings: [finding("F-001", "high", ["Task-1"])] } as never);
430
+ await restoring;
431
+ await __awaitReviewRoundForTests();
432
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
433
+ assert.equal(wakes.length, 1, "the high branch wakes despite all-undeterminable verdicts");
434
+ const ex = getExecution()!;
435
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
436
+ assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "undeterminable set still recorded");
437
+ await stopExecution(ctx, "teardown");
438
+ ctl.drainAll();
439
+ });
440
+
441
+ it("an unmapped high appends a plan task (proposed-task applied mechanically) and wakes once", async () => {
442
+ const { workdir, planPath, runId } = freshWorkdir();
443
+ const ctx = await startTerminal(planPath, workdir);
444
+ const ctl = controlledRunner();
445
+ __setAuditRunnerForTests(ctl.runner);
446
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
447
+ await tick();
448
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-009", "high", [], { proposedTask: "harden the retry budget guard" })] } as never);
449
+ await restoring;
450
+ await __awaitReviewRoundForTests();
451
+ const ex = getExecution()!;
452
+ const amended = ex.tasks.find((t) => t.id === "Task-3");
453
+ assert.ok(amended, "the unmapped high gained an appended task");
454
+ assert.equal(amended.status, "pending");
455
+ assert.match(amended.title, /fix F-009: harden the retry budget guard/);
456
+ assert.match(amended.title, /appended by execution review round 1/);
457
+ const planText = fs.readFileSync(planPath, "utf8");
458
+ assert.match(planText, /- `Task-3`: fix F-009: harden the retry budget guard/, "the plan file carries the appended bullet");
459
+ assert.match(planText, /## Verification Checks/, "the plan stays parseable (section intact)");
460
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
461
+ assert.ok(cp.execution?.tasks?.["Task-3"], "checkpoint carries the appended task");
462
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
463
+ assert.equal(wakes.length, 1);
464
+ assert.match(String(wakes[0].content), /Tasks appended to the plan for unmapped findings: Task-3/);
465
+ await stopExecution(ctx, "teardown");
466
+ ctl.drainAll();
467
+ });
468
+
469
+ it("completion with residual medium/low findings summarizes them in the completion message", async () => {
470
+ const { workdir, planPath, runId } = freshWorkdir();
471
+ const ctx = await startTerminal(planPath, workdir);
472
+ const ctl = controlledRunner();
473
+ __setAuditRunnerForTests(ctl.runner);
474
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
475
+ await tick();
476
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "clean", findings: [finding("F-002", "medium", []), finding("F-003", "low", ["Task-1"])] } as never);
477
+ await restoring;
478
+ await __awaitReviewRoundForTests();
479
+ assert.equal(getExecution(), null, "no high findings: the run completes");
480
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
481
+ const done = ctx.entries.filter((e) => e.customType === "pi-plans-complete");
482
+ assert.equal(done.length, 1);
483
+ assert.match(String(done[0].content), /Recorded findings that did not block completion: F-002 \(medium\), F-003 \(low\)/);
484
+ ctl.drainAll();
485
+ });
486
+
487
+ it("findings persist across the session snapshot and survive the fresh-budget renewal", async () => {
488
+ const { workdir, planPath } = freshWorkdir();
489
+ const ctx = await startTerminal(planPath, workdir);
490
+ // v0.9.3: this test drives rounds 1..5 on purpose — pin the 5-round
491
+ // budget (the no-UI default is now 3).
492
+ getExecution()!.reviewBudget = 5;
493
+ const ctl = controlledRunner();
494
+ __setAuditRunnerForTests(ctl.runner);
495
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
496
+ await tick();
497
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "r1", findings: [finding("F-001", "high", ["Task-2"])] } as never);
498
+ await restoring;
499
+ await __awaitReviewRoundForTests();
500
+ const ex = getExecution()!;
501
+ assert.equal(ex.audit.findings.length, 1, "findings in live state");
502
+ // The snapshot round-trip: a session restore rebuilds them.
503
+ const restored = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
504
+ await tick();
505
+ await restored;
506
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings survive the session restore");
507
+ // Renewal: /plans-execute grants a fresh budget and keeps the findings.
508
+ const { resumeActiveExecution } = await import("../src/exec.ts");
509
+ // Drive rounds 2-5 through the real path: fix, re-close, settle (restore
510
+ // is the settle entry in these tests) — the same high persists each time.
511
+ for (let r = 2; r <= 5; r++) {
512
+ for (const id of ["Task-1", "Task-2"]) {
513
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `fix round ${r}`);
514
+ }
515
+ persistTaskProgress(ctx);
516
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
517
+ await tick();
518
+ ctl.resolveRound({ round: r, passed: [], failed: [], undeterminable: [], report: `r${r}`, findings: [finding("F-001", "high", ["Task-2"])] } as never);
519
+ await settle;
520
+ await __awaitReviewRoundForTests();
521
+ }
522
+ // Round 5 committed with the high: the next terminal cycle hits the cap.
523
+ for (const id of ["Task-1", "Task-2"]) {
524
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", "post-cap close");
525
+ }
526
+ persistTaskProgress(ctx);
527
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
528
+ const paused = getExecution()!;
529
+ assert.equal(paused.stall.paused, true, "the cap pause fired");
530
+ assert.match(String(paused.stall.pausedReason), /high findings: F-001/, "the pause reason names the high findings");
531
+ assert.match(String(paused.stall.pausedReason), /execution review exhausted 5 rounds/, "the pause prefix phrase survives");
532
+ // Cancelled rounds burn no budget and loop nowhere — safe to leave the
533
+ // runner in this mode while checking the renewal semantics.
534
+ __setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
535
+ // v0.9.3: the resume is async and reports the outcome; this ctx has no
536
+ // panel, so the headless path re-grants the same budget (5).
537
+ const renewed = await resumeActiveExecution(ctx);
538
+ assert.ok(renewed.resumed, "renewal lifts the pause");
539
+ assert.equal(renewed.grantedBudget, 5, "the headless grant keeps the current budget");
540
+ assert.equal(getExecution()!.audit.rounds, 0, "fresh budget");
541
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "stable ids carry into the fresh budget");
542
+ ctl.drainAll();
543
+ await stopExecution(ctx, "teardown");
544
+ });
545
+ });
546
+
547
+ describe("mixed and hygiene rounds (v0.9.1 F-004/F-006/F-007/F-012)", () => {
548
+ it("a round with a failed check AND a high finding rolls back the union once and wakes exactly once (F-007)", async () => {
549
+ const { workdir, planPath, runId } = freshWorkdir();
550
+ const ctx = await startTerminal(planPath, workdir);
551
+ const ctl = controlledRunner();
552
+ __setAuditRunnerForTests(ctl.runner);
553
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
554
+ await tick();
555
+ // VC-001 fails (covers Task-1) and F-001 also maps to Task-1: the union
556
+ // must dedupe to one reopen of Task-1 plus Task-2 (VC-002 stays passed).
557
+ ctl.resolveRound({ round: 1, passed: ["VC-002"], failed: ["VC-001"], undeterminable: [], report: "mixed", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-1"], note: "n", evidence: "e", raw: "r" }] } as never);
558
+ await restoring;
559
+ await __awaitReviewRoundForTests();
560
+ const ex = getExecution()!;
561
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "Task-1 reopened once by both channels");
562
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "complete", "the passing check's task stays closed");
563
+ assert.equal(ex.items.find((i) => i.id === "VC-002")?.done, true, "unrelated pass kept");
564
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
565
+ assert.equal(wakes.length, 1, "one wake for the mixed round");
566
+ assert.match(String(wakes[0].content), /and failed checks: VC-001/);
567
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
568
+ await stopExecution(ctx, "teardown");
569
+ ctl.drainAll();
570
+ });
571
+
572
+ it("a pure VC-fail round keeps the v0.8 lead — never '0 high-severity findings' (F-004)", async () => {
573
+ const { workdir, planPath } = freshWorkdir();
574
+ const ctx = await startTerminal(planPath, workdir);
575
+ const ctl = controlledRunner();
576
+ __setAuditRunnerForTests(ctl.runner);
577
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
578
+ await tick();
579
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "vc2 broken" });
580
+ await restoring;
581
+ await __awaitReviewRoundForTests();
582
+ const wake = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
583
+ assert.equal(wake.length, 1);
584
+ assert.doesNotMatch(String(wake[0].content), /0 high-severity finding/);
585
+ assert.doesNotMatch(String(wake[0].content), /High findings:\n\(none/);
586
+ assert.match(String(wake[0].content), /round 1 failed\*\* — checks: VC-002/);
587
+ await stopExecution(ctx, "teardown");
588
+ ctl.drainAll();
589
+ });
590
+
591
+ it("the appended bullet carries its wave tail and sanitizes reviewer text (F-006/F-012)", async () => {
592
+ const { workdir, planPath } = freshWorkdir();
593
+ const ctx = await startTerminal(planPath, workdir);
594
+ const ctl = controlledRunner();
595
+ __setAuditRunnerForTests(ctl.runner);
596
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
597
+ await tick();
598
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden — the retry; budget guard", note: "n", evidence: "e", raw: "r" }] } as never);
599
+ await restoring;
600
+ await __awaitReviewRoundForTests();
601
+ const ex = getExecution()!;
602
+ const appended = ex.tasks.find((t) => t.id === "Task-3");
603
+ assert.ok(appended, "task appended");
604
+ const planText = fs.readFileSync(planPath, "utf8");
605
+ const bullet = planText.split("\n").find((l) => l.startsWith("- `Task-3`:"))!;
606
+ assert.match(bullet, /— wave: \d+$/, "the bullet carries the wave tail");
607
+ // Re-parse restores the same wave the live tree assigned (not wave 1).
608
+ const reparse = (await import("../src/plan.ts")).parsePlanTasks(planText);
609
+ const flat = reparse.tasks.flatMap(function walk(t: { children: unknown[] }) { return [t, ...t.children]; } as never) as never[];
610
+ const reparsed = flat.find((t: { id: string }) => t.id === "Task-3") as { wave: number; title: string; files: string[] };
611
+ assert.equal(reparsed.wave, appended.wave, "re-parse restores the live wave");
612
+ // Sanitized: em dash -> hyphen, ';' -> ',', no forged fields.
613
+ assert.ok(!/—|—/.test(reparsed.title.split("(appended")[0]), "em dashes sanitized out of the reviewer text");
614
+ assert.equal(reparsed.files.length, 0, "no fields forged from reviewer text");
615
+ await stopExecution(ctx, "teardown");
616
+ ctl.drainAll();
617
+ });
618
+ });
619
+
620
+ describe("no-report rounds preserve findings (v0.9.1 F-001)", () => {
621
+ const finding = (id: string) => ({ id, severity: "high" as const, taskIds: ["Task-2"], note: `${id} note`, evidence: "e", raw: "r" });
622
+
623
+ it("a spawn-failure round never vacuously completes a run with an unresolved high", async () => {
624
+ const { workdir, planPath } = freshWorkdir();
625
+ const ctx = await startTerminal(planPath, workdir);
626
+ const ctl = controlledRunner();
627
+ __setAuditRunnerForTests(ctl.runner);
628
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
629
+ await tick();
630
+ // Round 1: both VCs pass, one mapped high -> rollback + wake.
631
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
632
+ await restoring;
633
+ await __awaitReviewRoundForTests();
634
+ const wakesAfterR1 = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length;
635
+ // The executor fixes and re-closes; the settle starts round 2.
636
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
637
+ persistTaskProgress(ctx);
638
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
639
+ await tick();
640
+ // Round 2's subagent FAILS (null outcome): findings must be preserved,
641
+ // no completion, no second wake — the loop self-schedules. The inline
642
+ // settle chain stays pending through the self-scheduled round 3, so
643
+ // resolve round 3 BEFORE awaiting the settle.
644
+ ctl.resolveRound(null);
645
+ await tick();
646
+ const ex = getExecution()!;
647
+ assert.ok(ex, "a spawn-failure round never completes the run");
648
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "unresolved findings survive the no-report round");
649
+ assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length, wakesAfterR1, "no extra wake on the no-report round");
650
+ // Round 3 spawned (self-schedule) and reports the fix — now it completes.
651
+ ctl.resolveRound({ round: 3, passed: [], failed: [], undeterminable: [], report: "fixed", findings: [] } as never);
652
+ await settle;
653
+ await __awaitReviewRoundForTests();
654
+ assert.equal(getExecution(), null, "a clean re-report completes");
655
+ ctl.drainAll();
656
+ });
657
+
658
+ it("the two-consecutive-discard synthesis preserves findings instead of clearing them", async () => {
659
+ const { workdir, planPath } = freshWorkdir();
660
+ const ctx = await startTerminal(planPath, workdir);
661
+ const ctl = controlledRunner();
662
+ __setAuditRunnerForTests(ctl.runner);
663
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
664
+ await tick();
665
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
666
+ await restoring;
667
+ await __awaitReviewRoundForTests();
668
+ const engine = path.join(workdir, "lib", "engine.js");
669
+ const discard = () => {
670
+ fs.writeFileSync(engine, `export const engine = ${Math.random()};\n`, "utf8");
671
+ const later = new Date(Date.now() + 60_000);
672
+ fs.utimesSync(engine, later, later);
673
+ };
674
+ // Re-close, settle, then two fingerprint discards -> the synthesized
675
+ // commit must carry the previous findings forward, not wipe them.
676
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
677
+ persistTaskProgress(ctx);
678
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
679
+ await tick();
680
+ discard();
681
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
682
+ await tick();
683
+ discard();
684
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
685
+ await tick();
686
+ await tick();
687
+ const ex = getExecution()!;
688
+ assert.ok(ex, "the synthesis never completes the run");
689
+ assert.equal(ex.audit.rounds, 2, "the synthesis committed as round 2");
690
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings preserved through the discard synthesis");
691
+ await stopExecution(ctx, "teardown");
692
+ ctl.drainAll();
693
+ await settle;
694
+ await __awaitReviewRoundForTests();
695
+ });
696
+ });
697
+
698
+ describe("plan amendment re-stamps the checkpoint identity (v0.9.1 F-002)", () => {
699
+ it("an amended plan passes /resume-plans instead of plan-mismatch", async () => {
700
+ const { workdir, planPath, runId } = freshWorkdir();
701
+ const ctx = await startTerminal(planPath, workdir);
702
+ const ctl = controlledRunner();
703
+ __setAuditRunnerForTests(ctl.runner);
704
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
705
+ await tick();
706
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden the guard", note: "n", evidence: "e", raw: "r" }] } as never);
707
+ await restoring;
708
+ await __awaitReviewRoundForTests();
709
+ const ex = getExecution()!;
710
+ assert.ok(ex.tasks.find((t) => t.id === "Task-3"), "the amendment appended Task-3");
711
+ // The checkpoint identity now matches the AMENDED file, with provenance.
712
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
713
+ const { sha256File } = await import("../src/workflow-state.ts");
714
+ assert.equal(cp.plan?.sha256, sha256File(planPath), "identity re-stamped to the amended digest");
715
+ assert.equal(cp.execution?.planAmended?.round, 1);
716
+ assert.equal(cp.execution?.planAmended?.sha256, sha256File(planPath));
717
+ // A later resume accepts the amended plan (no plan-mismatch re-approval).
718
+ await stopExecution(ctx, "teardown");
719
+ ctl.drainAll();
720
+ const { loadExecutionFromCheckpoint } = await import("../src/exec.ts");
721
+ const result = loadExecutionFromCheckpoint(makeCtx(workdir), runId);
722
+ assert.equal(result.status, "loaded", `resume accepts the amended plan (${result.status})`);
723
+ assert.deepEqual(result.findings?.map((f) => f.id), ["F-009"], "the load result surfaces unresolved findings for the resume brief (F-005)");
724
+ await stopExecution(makeCtx(workdir), "post-check teardown");
725
+ });
726
+ });
727
+
728
+ describe("executor injection with findings (v0.9)", () => {
729
+ it("executionContextMessage lists unresolved high findings for the repairing agent", async () => {
730
+ const { workdir, planPath } = freshWorkdir();
731
+ const ctx = await startTerminal(planPath, workdir);
732
+ const ctl = controlledRunner();
733
+ __setAuditRunnerForTests(ctl.runner);
734
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
735
+ await tick();
736
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "loop misses union", evidence: "e", raw: "r" }] } as never);
737
+ await restoring;
738
+ await __awaitReviewRoundForTests();
739
+ const { executionContextMessage } = await import("../src/exec.ts");
740
+ const msg = executionContextMessage(ctx) ?? "";
741
+ assert.match(msg, /unresolved high-severity findings/);
742
+ assert.match(msg, /- F-001 \(Task-2\): loop misses union/);
743
+ assert.match(msg, /Fix them, then re-close the affected tasks/);
744
+ await stopExecution(ctx, "teardown");
745
+ ctl.drainAll();
746
+ });
747
+ });
748
+
749
+ /**
750
+ * v0.9.2 blocked-point escalation (RCA: a missed parent re-close left the tree
751
+ * non-terminal, the review could not start, and three silent no-op wakes
752
+ * paused the run with a watchdog metric instead of the blocker).
753
+ */
754
+ const PARENT_PLAN = `# PLAN_v1 - blocked fixture
755
+
756
+ ## Tasks
757
+
758
+ - Task-1: engine — files: lib/engine.js; wave: 1
759
+ - Task-2: host — files: lib/host.js; wave: 1
760
+ - Task-2.1: daemon wiring — files: lib/host.js
761
+
762
+ ## Verification Checks
763
+
764
+ - [ ] \`VC-001\` covers \`Task-1\`; pass condition: engine works; evidence: tests; metric: green.
765
+ - [ ] \`VC-002\` covers \`Task-2\`; pass condition: host works; evidence: tests; metric: green.
766
+ `;
767
+
768
+ describe("blocked-point escalation (v0.9.2)", () => {
769
+ let root = "";
770
+ let counter = 0;
771
+
772
+ before(() => {
773
+ root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-blocked-"));
774
+ });
775
+
776
+ after(() => {
777
+ fs.rmSync(root, { recursive: true, force: true });
778
+ __setAuditRunnerForTests(null);
779
+ });
780
+
781
+ function freshParentWorkdir(): { workdir: string; planPath: string; runId: string } {
782
+ counter += 1;
783
+ const workdir = path.join(root, `blocked-${counter}`);
784
+ fs.mkdirSync(path.join(workdir, "lib"), { recursive: true });
785
+ fs.writeFileSync(path.join(workdir, "lib", "engine.js"), "export const engine = 1;\n", "utf8");
786
+ fs.writeFileSync(path.join(workdir, "lib", "host.js"), "export const host = 1;\n", "utf8");
787
+ initState(workdir);
788
+ const { run } = startRun(workdir, { topic: `blocked${counter}`, skill: "plan-small", requestText: "demo" });
789
+ createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
790
+ const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
791
+ fs.mkdirSync(run.artifact_dir, { recursive: true });
792
+ fs.writeFileSync(planPath, PARENT_PLAN, "utf8");
793
+ mutateCheckpoint(workdir, run.run_id, (cp) =>
794
+ applyExecutionApproved(
795
+ applyPlanWritten({ ...cp, nextAction: "accept-execute" }, planIdentityOf(planPath, 1)),
796
+ { plan: planIdentityOf(planPath, 1), worktree: workdir, headAtApproval: null, approvedAt: cp.updatedAt },
797
+ ),
798
+ );
799
+ return { workdir, planPath, runId: run.run_id };
800
+ }
801
+
802
+ /** ctx + event driver (pattern: tests/exec.test.ts wireEvents, 677-690). */
803
+ function makeHarness(workdir: string) {
804
+ const entries: Array<{ customType: string; data?: unknown; content?: string; display?: boolean }> = [];
805
+ const ctx = {
806
+ cwd: workdir,
807
+ sessionManager: {},
808
+ hasUI: true,
809
+ mode: "tui",
810
+ entries,
811
+ ui: {
812
+ notify: () => {},
813
+ setStatus: () => {},
814
+ setWidget: () => {},
815
+ theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
816
+ },
817
+ isIdle: () => true,
818
+ hasPendingMessages: () => false,
819
+ } as never;
820
+ setMessagingApi({
821
+ appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }),
822
+ sendMessage: (message: { customType: string; content: string; display?: boolean }) => entries.push({ customType: message.customType, content: message.content, display: message.display }),
823
+ } as never);
824
+ const handlers = new Map<string, Array<(event: unknown, c: never) => unknown>>();
825
+ const ext = {
826
+ on: (name: string, fn: (event: unknown, c: never) => unknown) => {
827
+ const list = handlers.get(name) ?? [];
828
+ list.push(fn);
829
+ handlers.set(name, list);
830
+ },
831
+ } as never;
832
+ registerExecutionTurnHandlers(ext);
833
+ return {
834
+ ctx,
835
+ entries,
836
+ async fire(name: string, event: unknown = {}) {
837
+ for (const fn of handlers.get(name) ?? []) await fn(event, ctx);
838
+ },
839
+ /** One idle executor round that never touched a task or tool. */
840
+ async idleRound() {
841
+ await this.fire("agent_start");
842
+ await this.fire("turn_end", { message: { role: "assistant", stopReason: "stop", usage: { input: 1, output: 1 } } });
843
+ await this.fire("agent_settled");
844
+ },
845
+ };
846
+ }
847
+
848
+ it("names the missed parent, escalates before pausing, and clears on the parent re-close", async () => {
849
+ const { workdir, planPath, runId } = freshParentWorkdir();
850
+ const harness = makeHarness(workdir);
851
+ const ctx = harness.ctx;
852
+ await startExecution(ctx, { planPath, planTasks: parsePlanTasks(PARENT_PLAN), items: parseChecklist(PARENT_PLAN) });
853
+ // First pass: everything closed (children before the parent).
854
+ for (const id of ["Task-2.1", "Task-2", "Task-1"]) {
855
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
856
+ persistTaskProgress(ctx);
857
+ }
858
+ // Round 1 fails VC-002 -> Task-2 + Task-2.1 roll back.
859
+ const ctl = controlledRunner();
860
+ __setAuditRunnerForTests(ctl.runner);
861
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
862
+ await tick();
863
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "VC-002 fails" });
864
+ await restoring;
865
+ await __awaitReviewRoundForTests();
866
+ const ex = getExecution()!;
867
+ assert.equal(ex.audit.rounds, 1);
868
+ assert.deepEqual(ex.blocked?.rolledBack.sort(), ["Task-2", "Task-2.1"], "the authoritative rollback set is captured at commit time");
869
+ assert.deepEqual(ex.blocked?.tasks.sort(), ["Task-2", "Task-2.1"]);
870
+ assert.deepEqual(loadCheckpoint(workdir, runId).checkpoint.execution?.blocked?.tasks.sort(), ["Task-2", "Task-2.1"], "the blocker is persisted");
871
+ // The executor re-closes the CHILD only — the lattice-code shape.
872
+ applyTaskUpdate(ex.tasks, "Task-2.1", "complete", "wiring fixed");
873
+ persistTaskProgress(ctx);
874
+ assert.deepEqual(getExecution()!.blocked?.tasks, ["Task-2"], "closing the child does not close the parent");
875
+ assert.deepEqual(loadCheckpoint(workdir, runId).checkpoint.execution?.blocked?.tasks, ["Task-2"]);
876
+
877
+ // Blocked wake 1: escalated wording + one visible system line.
878
+ await harness.idleRound();
879
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 1);
880
+ const wake1 = harness.entries.filter((e) => e.customType === EXECUTION_CONTINUE_CUSTOM_TYPE).at(-1)?.content ?? "";
881
+ assert.match(wake1, /Review: NO round is running — the task tree is not terminal, so the review cannot start/);
882
+ assert.match(wake1, /BLOCKED — 1 task\(s\) reopened by round 1 are still open:/);
883
+ assert.match(wake1, /- Task-2 \(reopened by round 1\)/);
884
+ assert.match(wake1, /Closing a child does NOT close its parent/);
885
+ assert.match(wake1, /blocked wake 1\/3/);
886
+ const visible1 = harness.entries.filter((e) => e.customType === EXECUTION_BLOCKED_CUSTOM_TYPE);
887
+ assert.equal(visible1.length, 1, "one visible escalation line");
888
+ assert.equal(visible1[0]!.display, true);
889
+ assert.match(visible1[0]!.content ?? "", /execution blocked \(wake 1\/3\)/);
890
+
891
+ // Blocked wake 2: still no change -> escalate, still no pause.
892
+ await harness.idleRound();
893
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 2);
894
+ assert.equal(getExecution()!.stall.paused, false, "escalation precedes the pause");
895
+ assert.equal(harness.entries.filter((e) => e.customType === EXECUTION_BLOCKED_CUSTOM_TYPE).length, 2);
896
+ const wake2 = harness.entries.filter((e) => e.customType === EXECUTION_CONTINUE_CUSTOM_TYPE).at(-1)?.content ?? "";
897
+ assert.match(wake2, /blocked wake 2\/3/);
898
+
899
+ // Blocked wake 3: the watchdog pauses with the BLOCKER as the reason.
900
+ await harness.idleRound();
901
+ assert.equal(getExecution()!.stall.paused, true, "the third blocked wake pauses");
902
+ const reason = getExecution()!.stall.pausedReason ?? "";
903
+ assert.match(reason, /^blocked: review round 2 cannot start/);
904
+ assert.match(reason, /Task-2/);
905
+ assert.match(reason, /reopened by round 1/);
906
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.execution?.pausedReason, reason, "the pause reason is persisted");
907
+ // Levels 1 and 2 emit the visible line; the cap levels pauses instead of
908
+ // waking, and the pause carries the blocker in the pause row/notify.
909
+ assert.equal(harness.entries.filter((e) => e.customType === EXECUTION_BLOCKED_CUSTOM_TYPE).length, 2);
910
+ // The cap-pause prefix matching must not see a review-cap pause here.
911
+ await harness.fire("input", { source: "interactive" });
912
+ assert.equal(getExecution()!.stall.paused, false, "ordinary input resumes a blocked pause");
913
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 0, "a resume grants a fresh ladder");
914
+
915
+ // Re-closing the parent clears the blocker and the review starts by itself.
916
+ applyTaskUpdate(getExecution()!.tasks, "Task-2", "complete", "host fixed");
917
+ persistTaskProgress(ctx);
918
+ assert.equal(getExecution()!.blocked, null);
919
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.execution?.blocked, undefined, "the record is dropped, not left stale");
920
+ // The very next settled round launches round 2 by itself (turn_end owns
921
+ // the terminal-but-unaudited state — no wake, no user input).
922
+ const settled = harness.fire("turn_end", { message: { role: "assistant", stopReason: "stop", usage: { input: 1, output: 1 } } });
923
+ await tick();
924
+ assert.equal(ctl.calls(), 2, "round 2 spawned from the settle, not from a wake");
925
+ ctl.resolveRound({ round: 2, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
926
+ await settled;
927
+ await __awaitReviewRoundForTests();
928
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed", "the review ran once the tree became terminal");
929
+ __setAuditRunnerForTests(null);
930
+ ctl.drainAll();
931
+ });
932
+ it("real progress resets the ladder instead of hardening the wording (round-1 F-001)", async () => {
933
+ const { workdir, planPath, runId } = freshParentWorkdir();
934
+ const harness = makeHarness(workdir);
935
+ const ctx = harness.ctx;
936
+ await startExecution(ctx, { planPath, planTasks: parsePlanTasks(PARENT_PLAN), items: parseChecklist(PARENT_PLAN) });
937
+ for (const id of ["Task-2.1", "Task-2", "Task-1"]) {
938
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
939
+ persistTaskProgress(ctx);
940
+ }
941
+ const ctl = controlledRunner();
942
+ __setAuditRunnerForTests(ctl.runner);
943
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
944
+ await tick();
945
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "VC-002 fails" });
946
+ await restoring;
947
+ await __awaitReviewRoundForTests();
948
+ // Both reopened tasks stay open: one no-progress blocked wake.
949
+ await harness.idleRound();
950
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 1, "the first blocked wake counts as no progress");
951
+ assert.deepEqual(getExecution()!.blocked?.tasks.sort(), ["Task-2", "Task-2.1"]);
952
+ // The executor closes ONE of the two blockers -> real progress.
953
+ applyTaskUpdate(getExecution()!.tasks, "Task-2.1", "complete", "child fixed");
954
+ persistTaskProgress(ctx);
955
+ assert.deepEqual(getExecution()!.blocked?.tasks, ["Task-2"], "persistTaskProgress re-syncs the STORED list (the reason it cannot be the ladder baseline)");
956
+ await harness.idleRound();
957
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 0, "a shrunken blocker set resets the ladder");
958
+ const wake = harness.entries.filter((e) => e.customType === EXECUTION_CONTINUE_CUSTOM_TYPE).at(-1)?.content ?? "";
959
+ assert.doesNotMatch(wake, /blocked wake/, "the reset wake carries no escalation sentence");
960
+ assert.equal(harness.entries.filter((e) => e.customType === EXECUTION_BLOCKED_CUSTOM_TYPE).length, 1, "no new visible line for a reset");
961
+ // The fresh ladder then counts again: 1, 2, pause at 3.
962
+ await harness.idleRound();
963
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 1);
964
+ await harness.idleRound();
965
+ assert.equal(getExecution()!.blocked?.escalatedRounds, 2);
966
+ await harness.idleRound();
967
+ assert.equal(getExecution()!.stall.paused, true, "three consecutive no-progress wakes still pause");
968
+ assert.match(getExecution()!.stall.pausedReason ?? "", /blocked: review round 2 cannot start/);
969
+ assert.match(loadCheckpoint(workdir, runId).checkpoint.execution?.pausedReason ?? "", /Task-2/);
970
+ await stopExecution(ctx, "teardown");
971
+ ctl.drainAll();
972
+ });
973
+
974
+ it("the executor rules forbid waiting and require re-closing parents", async () => {
975
+ const { workdir, planPath } = freshParentWorkdir();
976
+ const harness = makeHarness(workdir);
977
+ await startExecution(harness.ctx, { planPath, planTasks: parsePlanTasks(PARENT_PLAN), items: parseChecklist(PARENT_PLAN) });
978
+ const msg = executionContextMessage(harness.ctx) ?? "";
979
+ assert.match(msg, /PARENTS INCLUDED/);
980
+ assert.match(msg, /Closing a task's children does NOT close the task/);
981
+ assert.match(msg, /NEVER wait for the review/);
982
+ await stopExecution(harness.ctx, "teardown");
983
+ });
984
+ });
985
+
986
+ /**
987
+ * Pre-feature checkpoint backfill (v0.9.2): the lattice-code run was paused by
988
+ * the old build, so its checkpoint has no `execution.blocked` record. The
989
+ * blocker is reconstructed from round 1's `audit.lastResult` + the plan's
990
+ * coverage — without re-applying the rollback (that would revert work the
991
+ * executor has since re-closed).
992
+ */
993
+ describe("pre-feature checkpoint backfill (v0.9.2)", () => {
994
+ it("reconstructs the real lattice-code blocker (Task-2, round 1) and persists it", async () => {
995
+ const fixtureDir = path.join(import.meta.dirname, "fixtures", "lattice-code-blocked");
996
+ const planText = fs.readFileSync(path.join(fixtureDir, "plan-v2-trimmed.md"), "utf8");
997
+ const state = JSON.parse(fs.readFileSync(path.join(fixtureDir, "state.json"), "utf8")) as {
998
+ tasks: Record<string, { status: string; evidence?: string; skipReason?: string }>;
999
+ audit: { rounds: number; lastResult: string; undeterminable: string[] };
1000
+ };
1001
+ const workdir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-backfill-"));
1002
+ spawnSync("git", ["init"], { cwd: workdir });
1003
+ spawnSync("git", ["config", "user.email", "t@example.com"], { cwd: workdir });
1004
+ spawnSync("git", ["config", "user.name", "T"], { cwd: workdir });
1005
+ fs.writeFileSync(path.join(workdir, "seed.txt"), "seed", "utf8");
1006
+ spawnSync("git", ["add", "-A"], { cwd: workdir });
1007
+ spawnSync("git", ["commit", "-m", "seed"], { cwd: workdir });
1008
+ initState(workdir);
1009
+ const { run } = startRun(workdir, { topic: "lattice-backfill", skill: "plan-small", requestText: "demo" });
1010
+ createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
1011
+ const planPath = path.join(run.artifact_dir, "PLAN_v2.md");
1012
+ fs.mkdirSync(run.artifact_dir, { recursive: true });
1013
+ fs.writeFileSync(planPath, planText, "utf8");
1014
+ mutateCheckpoint(workdir, run.run_id, (cp) => {
1015
+ const plan = planIdentityOf(planPath, 2);
1016
+ let next = applyPlanWritten({ ...cp, nextAction: "accept-execute" }, plan);
1017
+ next = applyExecutionApproved(next, {
1018
+ plan,
1019
+ worktree: workdir,
1020
+ headAtApproval: resolveHeadAt(workdir),
1021
+ approvedAt: next.updatedAt,
1022
+ });
1023
+ // The REAL slices; deliberately no `blocked` key (pre-feature checkpoint).
1024
+ return applyExecutionProgress(next, {
1025
+ tasks: state.tasks,
1026
+ audit: { rounds: state.audit.rounds, lastResult: state.audit.lastResult, undeterminable: state.audit.undeterminable },
1027
+ });
1028
+ });
1029
+ setRunStatus(workdir, run.run_id, "executing");
1030
+ const ctx = makeCtx(workdir, "tui");
1031
+ const load = loadExecutionFromCheckpoint(ctx, run.run_id);
1032
+ assert.equal(load.status, "loaded");
1033
+ assert.deepEqual(load.blocked?.tasks, ["Task-2"], "the real blocker is reconstructed");
1034
+ assert.equal(load.blocked?.round, 1);
1035
+ assert.ok(load.blocked?.rolledBack.includes("Task-7"), "coverage cascade covers the other tasks of the failed checks");
1036
+ assert.ok(load.blocked?.rolledBack.includes("Task-2.1"), "children cascade");
1037
+ const persisted = loadCheckpoint(workdir, run.run_id).checkpoint.execution?.blocked;
1038
+ assert.deepEqual(persisted?.tasks, ["Task-2"], "the reconstructed blocker is persisted");
1039
+ // The tree was NOT re-mutated: the reopened-then-re-closed tasks stay closed.
1040
+ const after = loadCheckpoint(workdir, run.run_id).checkpoint.execution?.tasks ?? {};
1041
+ assert.equal(after["Task-2"]?.status, "pending");
1042
+ assert.equal(after["Task-2.1"]?.status, "complete");
1043
+ assert.equal(after["Task-7"]?.status, "complete");
1044
+ assert.equal(getExecution()!.blocked?.tasks[0], "Task-2");
1045
+ // The /reload recovery path: a session snapshot written by a pre-feature
1046
+ // build carries the LIVE failed set (`audit.failed`) and no blocker
1047
+ // record — exactly the shape the stuck lattice-code session has.
1048
+ const live = getExecution()!;
1049
+ const snapshot = { ...live, audit: { ...live.audit, failed: state.audit.lastResult.split(",") } } as Record<string, unknown>;
1050
+ delete snapshot.blocked;
1051
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
1052
+ assert.deepEqual(getExecution()!.blocked?.tasks, ["Task-2"], "a pre-feature session snapshot is backfilled too");
1053
+ assert.equal(getExecution()!.blocked?.round, 1);
1054
+ await stopExecution(makeCtx(workdir), "teardown");
1055
+ });
1056
+ });
1057
+
1058
+ describe("execution-review budget (v0.9.3)", () => {
1059
+ const highFinding = (id: string, taskIds: string[] = ["Task-1"]) => ({
1060
+ id,
1061
+ severity: "high" as const,
1062
+ taskIds,
1063
+ note: `${id} note`,
1064
+ evidence: "lib/engine.js",
1065
+ raw: `- \` ${id}\``,
1066
+ });
1067
+
1068
+ /** A ctx whose budget menu answers with `choice` (a label prefix), or a
1069
+ * cancelled menu when null. UI.custom is either absent (menu path) or a
1070
+ * no-choice overlay (falls through to the menu). */
1071
+ function ctxWithBudgetMenu(workdir: string, choice: string | null, mode: "print" | "tui" = "tui", onAsk?: () => void) {
1072
+ const base = makeCtx(workdir, mode);
1073
+ const ui = (base as { ui: Record<string, unknown> }).ui;
1074
+ return {
1075
+ ...base,
1076
+ ui: {
1077
+ ...ui,
1078
+ select: async (_title: string, options: string[]): Promise<string | undefined> => {
1079
+ onAsk?.();
1080
+ if (choice === null) return undefined;
1081
+ return options.find((option) => option.startsWith(choice)) ?? options[0];
1082
+ },
1083
+ },
1084
+ } as typeof base;
1085
+ }
1086
+
1087
+ it("asks for the budget once, right before round 1, and persists the pick", async () => {
1088
+ const { workdir, planPath, runId } = freshWorkdir();
1089
+ await startTerminal(planPath, workdir);
1090
+ const ctl = controlledRunner();
1091
+ __setAuditRunnerForTests(ctl.runner);
1092
+ let asks = 0;
1093
+ const ctx = ctxWithBudgetMenu(workdir, "1 round", "tui", () => {
1094
+ asks += 1;
1095
+ });
1096
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1097
+ await tick();
1098
+ const ex = getExecution()!;
1099
+ assert.equal(ex.reviewBudget, 1, "the picked budget is live");
1100
+ assert.equal(ex.reviewBudgetDefaulted, false, "a user pick is not marked default");
1101
+ assert.ok(ex.review.inFlight, "the round spawns right after the pick");
1102
+ assert.equal(asks, 1, "the panel was asked exactly once");
1103
+ const cp = loadCheckpoint(workdir, runId);
1104
+ assert.ok(cp.status === "ok");
1105
+ assert.equal(cp.checkpoint.execution?.reviewBudget, 1, "the budget is persisted");
1106
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
1107
+ await restoring;
1108
+ await __awaitReviewRoundForTests();
1109
+ assert.equal(asks, 1, "no second ask for the same run");
1110
+ assert.equal(getExecution(), null, "the passing round completed the run");
1111
+ ctl.drainAll();
1112
+ __setAuditRunnerForTests(null);
1113
+ });
1114
+
1115
+ it("falls back to the default 3 with a visible note when no panel exists", async () => {
1116
+ const { workdir, planPath } = freshWorkdir();
1117
+ const ctx = await startTerminal(planPath, workdir);
1118
+ const ctl = controlledRunner();
1119
+ __setAuditRunnerForTests(ctl.runner);
1120
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1121
+ await tick();
1122
+ const ex = getExecution()!;
1123
+ assert.equal(ex.reviewBudget, 3, "the no-UI default applies");
1124
+ assert.equal(ex.reviewBudgetDefaulted, true, "and it is marked as the fallback");
1125
+ const notes = ctx.entries.filter((e) => e.customType === "pi-plans-review-budget-default");
1126
+ assert.equal(notes.length, 1, "one visible note, never silent");
1127
+ assert.match(String(notes[0].content), /default/);
1128
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
1129
+ await restoring;
1130
+ await __awaitReviewRoundForTests();
1131
+ ctl.drainAll();
1132
+ __setAuditRunnerForTests(null);
1133
+ });
1134
+
1135
+ it("completes at an exhausted budget with an unresolved high — no rollback, no append, no wake", async () => {
1136
+ const { workdir, planPath, runId } = freshWorkdir();
1137
+ const ctx = await startTerminal(planPath, workdir);
1138
+ // One round, then the budget is spent.
1139
+ getExecution()!.reviewBudget = 1;
1140
+ const ctl = controlledRunner();
1141
+ __setAuditRunnerForTests(ctl.runner);
1142
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1143
+ await tick();
1144
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "pass with a high", findings: [highFinding("F-001")] } as never);
1145
+ await restoring;
1146
+ await __awaitReviewRoundForTests();
1147
+ assert.equal(getExecution(), null, "the run completes at the exhausted budget");
1148
+ const cp = loadCheckpoint(workdir, runId);
1149
+ assert.ok(cp.status === "ok");
1150
+ assert.equal(cp.checkpoint.phase, "completed");
1151
+ assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length, 0, "no fix wake on the tolerated round");
1152
+ assert.equal(/Task-4/.test(fs.readFileSync(planPath, "utf8")), false, "no plan task was appended");
1153
+ const done = ctx.entries.filter((e) => e.customType === "pi-plans-complete");
1154
+ assert.equal(done.length, 1);
1155
+ assert.match(String(done[0].content), /review budget exhausted/, "the completion discloses the exhausted budget");
1156
+ assert.match(String(done[0].content), /F-001/, "and names the tolerated high finding");
1157
+ ctl.drainAll();
1158
+ __setAuditRunnerForTests(null);
1159
+ });
1160
+
1161
+ it("still wakes and repairs a high finding when the budget is NOT exhausted", async () => {
1162
+ const { workdir, planPath } = freshWorkdir();
1163
+ const ctx = await startTerminal(planPath, workdir);
1164
+ getExecution()!.reviewBudget = 3;
1165
+ const ctl = controlledRunner();
1166
+ __setAuditRunnerForTests(ctl.runner);
1167
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1168
+ await tick();
1169
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "pass with a high", findings: [highFinding("F-001")] } as never);
1170
+ await restoring;
1171
+ await __awaitReviewRoundForTests();
1172
+ const ex = getExecution();
1173
+ assert.notEqual(ex, null, "the run does not complete with a live high inside the budget");
1174
+ assert.equal(ex!.tasks.find((t) => t.id === "Task-1")?.status, "pending", "the mapped task rolled back");
1175
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
1176
+ assert.equal(wakes.length, 1, "exactly one repair wake");
1177
+ ctl.drainAll();
1178
+ await stopExecution(ctx, "teardown");
1179
+ __setAuditRunnerForTests(null);
1180
+ });
1181
+
1182
+ it("unlimited: three identical no-report rounds trip the no-progress valve", async () => {
1183
+ const { workdir, planPath } = freshWorkdir();
1184
+ const ctx = await startTerminal(planPath, workdir);
1185
+ getExecution()!.reviewBudget = "unlimited";
1186
+ // A spawn failure carries no report: the signature must still advance
1187
+ // (round-1 F-002) — otherwise the valve could never catch a dead runner.
1188
+ __setAuditRunnerForTests(async () => null);
1189
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1190
+ const ex = getExecution()!;
1191
+ assert.equal(ex.audit.rounds, 3, "three rounds committed before the valve");
1192
+ assert.equal(ex.reviewRoundsTotal, 3, "the cumulative counter agrees");
1193
+ assert.equal(ex.stall.paused, true, "the valve pauses the run");
1194
+ assert.match(ex.stall.pausedReason ?? "", /^execution review stalled/);
1195
+ assert.match(ex.stall.pausedReason ?? "", /3 consecutive rounds/);
1196
+ assert.equal(ex.reviewNoProgress?.streak, 3);
1197
+ assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the pause is in-band");
1198
+ await stopExecution(ctx, "teardown");
1199
+ __setAuditRunnerForTests(null);
1200
+ });
1201
+
1202
+ it("unlimited: the run-cumulative hard cap pauses at 50 rounds", async () => {
1203
+ const { workdir, planPath } = freshWorkdir();
1204
+ const ctx = await startTerminal(planPath, workdir);
1205
+ getExecution()!.reviewBudget = "unlimited";
1206
+ getExecution()!.reviewRoundsTotal = 49;
1207
+ __setAuditRunnerForTests(async () => null);
1208
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1209
+ const ex = getExecution()!;
1210
+ assert.equal(ex.reviewRoundsTotal, 50, "the cap counts the whole run");
1211
+ assert.equal(ex.stall.paused, true);
1212
+ assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 50 rounds/);
1213
+ assert.match(ex.stall.pausedReason ?? "", /unlimited budget safety cap/);
1214
+ await stopExecution(ctx, "teardown");
1215
+ __setAuditRunnerForTests(null);
1216
+ });
1217
+
1218
+ it("the at-pause grant re-asks the budget; Esc keeps the pause, an unlimited pick adds 50", async () => {
1219
+ const { workdir, planPath } = freshWorkdir();
1220
+ const ctx = await startTerminal(planPath, workdir);
1221
+ getExecution()!.reviewBudget = "unlimited";
1222
+ getExecution()!.reviewRoundsTotal = 49;
1223
+ __setAuditRunnerForTests(async () => null);
1224
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1225
+ assert.equal(getExecution()!.stall.paused, true, "paused at the hard cap");
1226
+ const { resumeActiveExecution } = await import("../src/exec.ts");
1227
+ // Esc (cancelled menu) keeps the pause — never a silent grant.
1228
+ const declined = await resumeActiveExecution(ctxWithBudgetMenu(workdir, null));
1229
+ assert.equal(declined.resumed, false);
1230
+ assert.equal(declined.budgetDeclined, true);
1231
+ assert.equal(getExecution()!.stall.paused, true, "the pause stands after Esc");
1232
+ // Picking unlimited lifts the cap by exactly one window and resets the
1233
+ // per-grant counters — the cumulative total never resets.
1234
+ const ctl = controlledRunner();
1235
+ __setAuditRunnerForTests(ctl.runner);
1236
+ const grantingCtx = ctxWithBudgetMenu(workdir, "unlimited");
1237
+ const granted = await resumeActiveExecution(grantingCtx);
1238
+ assert.equal(granted.resumed, true);
1239
+ assert.equal(granted.grantedBudget, "unlimited");
1240
+ const ex = getExecution()!;
1241
+ assert.equal(ex.reviewCapExtension, 50, "the unlimited grant extends the hard cap");
1242
+ assert.equal(ex.audit.rounds, 0, "the per-grant round counter resets");
1243
+ assert.equal(ex.reviewRoundsTotal, 50, "the run-cumulative counter does not");
1244
+ assert.equal(ex.reviewNoProgress, undefined, "the valve window restarts");
1245
+ assert.equal(ex.stall.paused, false, "the run resumed");
1246
+ assert.ok(
1247
+ grantingCtx.entries.some((e) => e.customType === "pi-plans-review-budget-granted" && /hard cap now 100 rounds/.test(String(e.content))),
1248
+ "the grant names the lifted cap",
1249
+ );
1250
+ ctl.drainAll();
1251
+ await stopExecution(ctx, "teardown");
1252
+ __setAuditRunnerForTests(null);
1253
+ });
1254
+
1255
+ it("a headless session keeps the default budget and can still be granted at a pause (round-1 F-001)", async () => {
1256
+ const { workdir, planPath, runId } = freshWorkdir();
1257
+ const ctx = await startTerminal(planPath, workdir);
1258
+ // The SDK's ui.select is REQUIRED, so a real headless context carries a
1259
+ // function that cannot ask. This host throws if the code ever calls it.
1260
+ const headlessCtx = {
1261
+ ...ctx,
1262
+ hasUI: false,
1263
+ ui: {
1264
+ ...(ctx as { ui: Record<string, unknown> }).ui,
1265
+ select: async (): Promise<string | undefined> => {
1266
+ throw new Error("headless select must never be called");
1267
+ },
1268
+ },
1269
+ } as typeof ctx;
1270
+ const ctl = controlledRunner();
1271
+ __setAuditRunnerForTests(ctl.runner);
1272
+ const restoring = restoreFromSession(headlessCtx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1273
+ await tick();
1274
+ const ex = getExecution()!;
1275
+ assert.equal(ex.reviewBudget, 3, "a headless session gets the default budget");
1276
+ assert.equal(ex.reviewBudgetDefaulted, true, "and it is marked as the fallback");
1277
+ assert.ok(ex.review.inFlight, "the round spawned without asking");
1278
+ const cp = loadCheckpoint(workdir, runId);
1279
+ assert.ok(cp.status === "ok");
1280
+ assert.equal(cp.checkpoint.execution?.reviewBudget, 3, "the default is persisted");
1281
+ // A cancelled round ends the chain without burning budget or
1282
+ // self-scheduling; the next restore then sees an exhausted budget.
1283
+ ctl.resolveRound({ cancelled: true } as never);
1284
+ await restoring;
1285
+ await __awaitReviewRoundForTests();
1286
+ const live = getExecution()!;
1287
+ live.audit.rounds = 3;
1288
+ live.reviewRoundsTotal = 3;
1289
+ live.audit.failed = ["VC-001"];
1290
+ await restoreFromSession(headlessCtx, [{ type: "custom", customType: "pi-plans-exec", data: live }]);
1291
+ assert.equal(getExecution()!.stall.paused, true, "an owed review at an exhausted budget pauses");
1292
+ const { resumeActiveExecution } = await import("../src/exec.ts");
1293
+ const granted = await resumeActiveExecution(headlessCtx);
1294
+ assert.equal(granted.resumed, true, "a headless grant is not declined");
1295
+ assert.equal(granted.budgetDeclined, undefined, "no panel decline path in a headless session");
1296
+ assert.equal(granted.grantedBudget, 3, "the headless grant keeps the current budget");
1297
+ assert.equal(getExecution()!.stall.paused, false, "the run resumes instead of staying paused forever");
1298
+ assert.equal(getExecution()!.audit.rounds, 0, "and the per-grant counter reset");
1299
+ ctl.resolveRound({ cancelled: true } as never);
1300
+ await __awaitReviewRoundForTests();
1301
+ await stopExecution(headlessCtx, "teardown");
1302
+ __setAuditRunnerForTests(null);
1303
+ });
1304
+
1305
+ it("an auto-approve session never asks for the budget (recorded plan decision)", async () => {
1306
+ const { workdir, planPath } = freshWorkdir();
1307
+ const ctx = await startTerminal(planPath, workdir);
1308
+ // A TUI-capable host: without the auto-approve gate this would ask.
1309
+ const answering = ctxWithBudgetMenu(workdir, "1 round", "tui", () => {
1310
+ throw new Error("auto-approve must not ask for the budget");
1311
+ });
1312
+ const previous = process.env.PI_PLANS_AUTO_APPROVE;
1313
+ process.env.PI_PLANS_AUTO_APPROVE = "1";
1314
+ try {
1315
+ const ctl = controlledRunner();
1316
+ __setAuditRunnerForTests(ctl.runner);
1317
+ const restoring = restoreFromSession(answering, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
1318
+ await tick();
1319
+ assert.equal(getExecution()!.reviewBudget, 3, "auto-approve takes the default budget");
1320
+ assert.equal(getExecution()!.reviewBudgetDefaulted, true);
1321
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
1322
+ await restoring;
1323
+ await __awaitReviewRoundForTests();
1324
+ ctl.drainAll();
1325
+ } finally {
1326
+ if (previous === undefined) delete process.env.PI_PLANS_AUTO_APPROVE;
1327
+ else process.env.PI_PLANS_AUTO_APPROVE = previous;
1328
+ }
1329
+ __setAuditRunnerForTests(null);
1330
+ await stopExecution(ctx, "teardown");
1331
+ });
1332
+
1333
+ it("the budget counters and the valve state survive a session restore", async () => {
1334
+ const { workdir, planPath } = freshWorkdir();
1335
+ const ctx = await startTerminal(planPath, workdir);
1336
+ const live = getExecution()!;
1337
+ live.reviewBudget = "unlimited";
1338
+ live.reviewRoundsTotal = 7;
1339
+ live.reviewCapExtension = 50;
1340
+ live.reviewNoProgress = { key: "VC-002|VC-001|", streak: 2 };
1341
+ // A cancelled round neither burns the budget nor mutates the counters.
1342
+ __setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
1343
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: live }]);
1344
+ const ex = getExecution()!;
1345
+ assert.equal(ex.reviewBudget, "unlimited", "the budget survives");
1346
+ assert.equal(ex.reviewRoundsTotal, 7, "the cumulative counter survives (no free window)");
1347
+ assert.equal(ex.reviewCapExtension, 50, "the granted extension survives");
1348
+ assert.deepEqual(ex.reviewNoProgress, { key: "VC-002|VC-001|", streak: 2 }, "the valve streak survives");
1349
+ await stopExecution(ctx, "teardown");
1350
+ __setAuditRunnerForTests(null);
1351
+ });
1352
+ });