pi-plans 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -72,7 +72,7 @@ function freshWorkdir(): { workdir: string; planPath: string; runId: string } {
72
72
  return { workdir, planPath, runId: run.run_id };
73
73
  }
74
74
 
75
- function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }) {
75
+ function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }, components?: RefineOverlayComponent[]) {
76
76
  const entries: Array<{ customType: string; data?: unknown; content?: string }> = [];
77
77
  const ctx = {
78
78
  cwd: workdir,
@@ -85,9 +85,12 @@ function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?:
85
85
  setStatus: () => {},
86
86
  setWidget: () => {},
87
87
  theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
88
- // Minimal overlay host: counts ui.custom opens (one per controller).
89
- custom: () => {
88
+ // Minimal overlay host: counts ui.custom opens (one per controller)
89
+ // and captures the rendered component so tests can drive its input.
90
+ custom: (render: (tui: unknown, theme: unknown, kb: unknown, done: () => void) => { handleInput(data: string): void }) => {
90
91
  if (customOpens) customOpens.count += 1;
92
+ const component = render({ requestRender() {}, terminal: undefined }, { fg: (_c: string, t: string) => t, bold: (t: string) => t }, undefined, () => {});
93
+ if (components) components.push(component);
91
94
  return Promise.resolve();
92
95
  },
93
96
  },
@@ -296,14 +299,19 @@ describe("execution-review loop (v0.8)", () => {
296
299
  const ctl = controlledRunner();
297
300
  __setAuditRunnerForTests(ctl.runner);
298
301
  const opens = { count: 0 };
299
- const ctx = makeCtx(workdir, "tui", opens);
302
+ const components: Array<{ handleInput(data: string): void }> = [];
303
+ const ctx = makeCtx(workdir, "tui", opens, components);
300
304
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
301
305
  assert.equal(opens.count, 1, "the round opened its overlay before spawning");
302
- // ESC closed the (one-shot) controller; the reopen shortcut rebuilds it
303
- // from engine-held lane state while the round is still in flight.
306
+ // ESC closed the (one-shot) controller — only THEN may reopen rebuild it
307
+ // (the anti-stacking guard keeps a second overlay off a live one).
308
+ components[0]!.handleInput("\x1b");
304
309
  const { reopenReviewOverlay } = await import("../src/exec.ts");
305
310
  reopenReviewOverlay(ctx);
306
- assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round");
311
+ assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round after ESC");
312
+ // A second reopen while the new controller is live must NOT stack.
313
+ reopenReviewOverlay(ctx);
314
+ assert.equal(opens.count, 2, "reopen never stacks a second live overlay");
307
315
  ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "done" });
308
316
  await __awaitReviewRoundForTests();
309
317
  // No round in flight → the shortcut is inert.
@@ -329,3 +337,388 @@ describe("execution-review loop (v0.8)", () => {
329
337
  await stopExecution(makeCtx(workdir), "teardown");
330
338
  });
331
339
  });
340
+
341
+ describe("findings-driven fix loop (v0.9)", () => {
342
+ const finding = (id: string, severity: "high" | "medium" | "low", taskIds: string[], extra: Partial<{ note: string; proposedTask: string }> = {}) => ({
343
+ id, severity, taskIds, note: extra.note ?? `${id} note`, evidence: "src/lib", raw: `- \` ${id}\` raw`,
344
+ ...(extra.proposedTask ? { proposedTask: extra.proposedTask } : {}),
345
+ });
346
+
347
+ it("all VCs pass but a mapped high finding blocks completion: rollback + exactly one wake + NOT done", async () => {
348
+ const { workdir, planPath, runId } = freshWorkdir();
349
+ const ctx = await startTerminal(planPath, workdir);
350
+ const ctl = controlledRunner();
351
+ __setAuditRunnerForTests(ctl.runner);
352
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
353
+ await tick();
354
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "ok but findings", findings: [finding("F-001", "high", ["Task-2"])] } as never);
355
+ await restoring;
356
+ await __awaitReviewRoundForTests();
357
+ const ex = getExecution()!;
358
+ assert.ok(ex, "high findings never complete the run (liveness)");
359
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
360
+ assert.equal(getRun(workdir, runId)?.status, "executing", "back to executing for the fix round");
361
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending", "the high finding's mapped task rolled back");
362
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
363
+ assert.equal(wakes.length, 1, "exactly one wake");
364
+ assert.match(String(wakes[0].content), /1 high-severity finding/);
365
+ assert.match(String(wakes[0].content), /F-001/);
366
+ assert.match(String(wakes[0].content), /Full round report: /, "the wake references the round report path");
367
+ await stopExecution(ctx, "teardown");
368
+ ctl.drainAll();
369
+ });
370
+
371
+ it("keep-done asymmetry: a pure-high rollback keeps earlier VC passes; a VC-fail rollback invalidates", async () => {
372
+ const { workdir, planPath } = freshWorkdir();
373
+ const ctx = await startTerminal(planPath, workdir);
374
+ const ctl = controlledRunner();
375
+ __setAuditRunnerForTests(ctl.runner);
376
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
377
+ await tick();
378
+ // Round 1: VC-001 passes, VC-002 passes, but F-001 (high) maps to Task-1 —
379
+ // which VC-001 covers. The high rollback must NOT clear VC-001's done.
380
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001", "high", ["Task-1"])] } as never);
381
+ await restoring;
382
+ await __awaitReviewRoundForTests();
383
+ const ex = getExecution()!;
384
+ assert.equal(ex.items.find((i) => i.id === "VC-001")?.done, true, "pure-high rollback keeps the earlier pass");
385
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
386
+ await stopExecution(ctx, "teardown");
387
+ ctl.drainAll();
388
+
389
+ // Contrast: a VC-fail rollback invalidates checks covering the reopened task.
390
+ const second = freshWorkdir();
391
+ const ctx2 = await startTerminal(second.planPath, second.workdir);
392
+ const ctl2 = controlledRunner();
393
+ __setAuditRunnerForTests(ctl2.runner);
394
+ const restoring2 = restoreFromSession(ctx2, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
395
+ await tick();
396
+ ctl2.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "r1" });
397
+ await restoring2;
398
+ await __awaitReviewRoundForTests();
399
+ const ex2 = getExecution()!;
400
+ assert.equal(ex2.items.find((i) => i.id === "VC-001")?.done, true, "unrelated pass kept");
401
+ assert.equal(ex2.tasks.find((t) => t.id === "Task-2")?.status, "pending", "failed check's task rolled back");
402
+ await stopExecution(ctx2, "teardown");
403
+ ctl2.drainAll();
404
+ });
405
+
406
+ it("undeterminable round carrying a high finding still wakes (high wins over self-schedule)", async () => {
407
+ const { workdir, planPath } = freshWorkdir();
408
+ const ctx = await startTerminal(planPath, workdir);
409
+ const ctl = controlledRunner();
410
+ __setAuditRunnerForTests(ctl.runner);
411
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
412
+ await tick();
413
+ ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "unreadable", findings: [finding("F-001", "high", ["Task-1"])] } as never);
414
+ await restoring;
415
+ await __awaitReviewRoundForTests();
416
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
417
+ assert.equal(wakes.length, 1, "the high branch wakes despite all-undeterminable verdicts");
418
+ const ex = getExecution()!;
419
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "mapped task reopened");
420
+ assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "undeterminable set still recorded");
421
+ await stopExecution(ctx, "teardown");
422
+ ctl.drainAll();
423
+ });
424
+
425
+ it("an unmapped high appends a plan task (proposed-task applied mechanically) and wakes once", async () => {
426
+ const { workdir, planPath, runId } = freshWorkdir();
427
+ const ctx = await startTerminal(planPath, workdir);
428
+ const ctl = controlledRunner();
429
+ __setAuditRunnerForTests(ctl.runner);
430
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
431
+ await tick();
432
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-009", "high", [], { proposedTask: "harden the retry budget guard" })] } as never);
433
+ await restoring;
434
+ await __awaitReviewRoundForTests();
435
+ const ex = getExecution()!;
436
+ const amended = ex.tasks.find((t) => t.id === "Task-3");
437
+ assert.ok(amended, "the unmapped high gained an appended task");
438
+ assert.equal(amended.status, "pending");
439
+ assert.match(amended.title, /fix F-009: harden the retry budget guard/);
440
+ assert.match(amended.title, /appended by execution review round 1/);
441
+ const planText = fs.readFileSync(planPath, "utf8");
442
+ assert.match(planText, /- `Task-3`: fix F-009: harden the retry budget guard/, "the plan file carries the appended bullet");
443
+ assert.match(planText, /## Verification Checks/, "the plan stays parseable (section intact)");
444
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
445
+ assert.ok(cp.execution?.tasks?.["Task-3"], "checkpoint carries the appended task");
446
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
447
+ assert.equal(wakes.length, 1);
448
+ assert.match(String(wakes[0].content), /Tasks appended to the plan for unmapped findings: Task-3/);
449
+ await stopExecution(ctx, "teardown");
450
+ ctl.drainAll();
451
+ });
452
+
453
+ it("completion with residual medium/low findings summarizes them in the completion message", async () => {
454
+ const { workdir, planPath, runId } = freshWorkdir();
455
+ const ctx = await startTerminal(planPath, workdir);
456
+ const ctl = controlledRunner();
457
+ __setAuditRunnerForTests(ctl.runner);
458
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
459
+ await tick();
460
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "clean", findings: [finding("F-002", "medium", []), finding("F-003", "low", ["Task-1"])] } as never);
461
+ await restoring;
462
+ await __awaitReviewRoundForTests();
463
+ assert.equal(getExecution(), null, "no high findings: the run completes");
464
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
465
+ const done = ctx.entries.filter((e) => e.customType === "pi-plans-complete");
466
+ assert.equal(done.length, 1);
467
+ assert.match(String(done[0].content), /Recorded findings that did not block completion: F-002 \(medium\), F-003 \(low\)/);
468
+ ctl.drainAll();
469
+ });
470
+
471
+ it("findings persist across the session snapshot and survive the fresh-budget renewal", async () => {
472
+ const { workdir, planPath } = freshWorkdir();
473
+ const ctx = await startTerminal(planPath, workdir);
474
+ const ctl = controlledRunner();
475
+ __setAuditRunnerForTests(ctl.runner);
476
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
477
+ await tick();
478
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "r1", findings: [finding("F-001", "high", ["Task-2"])] } as never);
479
+ await restoring;
480
+ await __awaitReviewRoundForTests();
481
+ const ex = getExecution()!;
482
+ assert.equal(ex.audit.findings.length, 1, "findings in live state");
483
+ // The snapshot round-trip: a session restore rebuilds them.
484
+ const restored = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
485
+ await tick();
486
+ await restored;
487
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings survive the session restore");
488
+ // Renewal: /plans-execute grants a fresh budget and keeps the findings.
489
+ const { resumeActiveExecution } = await import("../src/exec.ts");
490
+ // Drive rounds 2-5 through the real path: fix, re-close, settle (restore
491
+ // is the settle entry in these tests) — the same high persists each time.
492
+ for (let r = 2; r <= 5; r++) {
493
+ for (const id of ["Task-1", "Task-2"]) {
494
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `fix round ${r}`);
495
+ }
496
+ persistTaskProgress(ctx);
497
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
498
+ await tick();
499
+ ctl.resolveRound({ round: r, passed: [], failed: [], undeterminable: [], report: `r${r}`, findings: [finding("F-001", "high", ["Task-2"])] } as never);
500
+ await settle;
501
+ await __awaitReviewRoundForTests();
502
+ }
503
+ // Round 5 committed with the high: the next terminal cycle hits the cap.
504
+ for (const id of ["Task-1", "Task-2"]) {
505
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", "post-cap close");
506
+ }
507
+ persistTaskProgress(ctx);
508
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
509
+ const paused = getExecution()!;
510
+ assert.equal(paused.stall.paused, true, "the cap pause fired");
511
+ assert.match(String(paused.stall.pausedReason), /high findings: F-001/, "the pause reason names the high findings");
512
+ assert.match(String(paused.stall.pausedReason), /execution review exhausted 5 rounds/, "the pause prefix phrase survives");
513
+ // Cancelled rounds burn no budget and loop nowhere — safe to leave the
514
+ // runner in this mode while checking the renewal semantics.
515
+ __setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
516
+ assert.ok(resumeActiveExecution(ctx), "renewal lifts the pause");
517
+ assert.equal(getExecution()!.audit.rounds, 0, "fresh budget");
518
+ assert.deepEqual(getExecution()!.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "stable ids carry into the fresh budget");
519
+ ctl.drainAll();
520
+ await stopExecution(ctx, "teardown");
521
+ });
522
+ });
523
+
524
+ describe("mixed and hygiene rounds (v0.9.1 F-004/F-006/F-007/F-012)", () => {
525
+ it("a round with a failed check AND a high finding rolls back the union once and wakes exactly once (F-007)", async () => {
526
+ const { workdir, planPath, runId } = freshWorkdir();
527
+ const ctx = await startTerminal(planPath, workdir);
528
+ const ctl = controlledRunner();
529
+ __setAuditRunnerForTests(ctl.runner);
530
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
531
+ await tick();
532
+ // VC-001 fails (covers Task-1) and F-001 also maps to Task-1: the union
533
+ // must dedupe to one reopen of Task-1 plus Task-2 (VC-002 stays passed).
534
+ ctl.resolveRound({ round: 1, passed: ["VC-002"], failed: ["VC-001"], undeterminable: [], report: "mixed", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-1"], note: "n", evidence: "e", raw: "r" }] } as never);
535
+ await restoring;
536
+ await __awaitReviewRoundForTests();
537
+ const ex = getExecution()!;
538
+ assert.equal(ex.tasks.find((t) => t.id === "Task-1")?.status, "pending", "Task-1 reopened once by both channels");
539
+ assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "complete", "the passing check's task stays closed");
540
+ assert.equal(ex.items.find((i) => i.id === "VC-002")?.done, true, "unrelated pass kept");
541
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
542
+ assert.equal(wakes.length, 1, "one wake for the mixed round");
543
+ assert.match(String(wakes[0].content), /and failed checks: VC-001/);
544
+ assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing");
545
+ await stopExecution(ctx, "teardown");
546
+ ctl.drainAll();
547
+ });
548
+
549
+ it("a pure VC-fail round keeps the v0.8 lead — never '0 high-severity findings' (F-004)", async () => {
550
+ const { workdir, planPath } = freshWorkdir();
551
+ const ctx = await startTerminal(planPath, workdir);
552
+ const ctl = controlledRunner();
553
+ __setAuditRunnerForTests(ctl.runner);
554
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
555
+ await tick();
556
+ ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "vc2 broken" });
557
+ await restoring;
558
+ await __awaitReviewRoundForTests();
559
+ const wake = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
560
+ assert.equal(wake.length, 1);
561
+ assert.doesNotMatch(String(wake[0].content), /0 high-severity finding/);
562
+ assert.doesNotMatch(String(wake[0].content), /High findings:\n\(none/);
563
+ assert.match(String(wake[0].content), /round 1 failed\*\* — checks: VC-002/);
564
+ await stopExecution(ctx, "teardown");
565
+ ctl.drainAll();
566
+ });
567
+
568
+ it("the appended bullet carries its wave tail and sanitizes reviewer text (F-006/F-012)", async () => {
569
+ const { workdir, planPath } = freshWorkdir();
570
+ const ctx = await startTerminal(planPath, workdir);
571
+ const ctl = controlledRunner();
572
+ __setAuditRunnerForTests(ctl.runner);
573
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
574
+ await tick();
575
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden — the retry; budget guard", note: "n", evidence: "e", raw: "r" }] } as never);
576
+ await restoring;
577
+ await __awaitReviewRoundForTests();
578
+ const ex = getExecution()!;
579
+ const appended = ex.tasks.find((t) => t.id === "Task-3");
580
+ assert.ok(appended, "task appended");
581
+ const planText = fs.readFileSync(planPath, "utf8");
582
+ const bullet = planText.split("\n").find((l) => l.startsWith("- `Task-3`:"))!;
583
+ assert.match(bullet, /— wave: \d+$/, "the bullet carries the wave tail");
584
+ // Re-parse restores the same wave the live tree assigned (not wave 1).
585
+ const reparse = (await import("../src/plan.ts")).parsePlanTasks(planText);
586
+ const flat = reparse.tasks.flatMap(function walk(t: { children: unknown[] }) { return [t, ...t.children]; } as never) as never[];
587
+ const reparsed = flat.find((t: { id: string }) => t.id === "Task-3") as { wave: number; title: string; files: string[] };
588
+ assert.equal(reparsed.wave, appended.wave, "re-parse restores the live wave");
589
+ // Sanitized: em dash -> hyphen, ';' -> ',', no forged fields.
590
+ assert.ok(!/—|—/.test(reparsed.title.split("(appended")[0]), "em dashes sanitized out of the reviewer text");
591
+ assert.equal(reparsed.files.length, 0, "no fields forged from reviewer text");
592
+ await stopExecution(ctx, "teardown");
593
+ ctl.drainAll();
594
+ });
595
+ });
596
+
597
+ describe("no-report rounds preserve findings (v0.9.1 F-001)", () => {
598
+ const finding = (id: string) => ({ id, severity: "high" as const, taskIds: ["Task-2"], note: `${id} note`, evidence: "e", raw: "r" });
599
+
600
+ it("a spawn-failure round never vacuously completes a run with an unresolved high", async () => {
601
+ const { workdir, planPath } = freshWorkdir();
602
+ const ctx = await startTerminal(planPath, workdir);
603
+ const ctl = controlledRunner();
604
+ __setAuditRunnerForTests(ctl.runner);
605
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
606
+ await tick();
607
+ // Round 1: both VCs pass, one mapped high -> rollback + wake.
608
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
609
+ await restoring;
610
+ await __awaitReviewRoundForTests();
611
+ const wakesAfterR1 = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length;
612
+ // The executor fixes and re-closes; the settle starts round 2.
613
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
614
+ persistTaskProgress(ctx);
615
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
616
+ await tick();
617
+ // Round 2's subagent FAILS (null outcome): findings must be preserved,
618
+ // no completion, no second wake — the loop self-schedules. The inline
619
+ // settle chain stays pending through the self-scheduled round 3, so
620
+ // resolve round 3 BEFORE awaiting the settle.
621
+ ctl.resolveRound(null);
622
+ await tick();
623
+ const ex = getExecution()!;
624
+ assert.ok(ex, "a spawn-failure round never completes the run");
625
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "unresolved findings survive the no-report round");
626
+ assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed").length, wakesAfterR1, "no extra wake on the no-report round");
627
+ // Round 3 spawned (self-schedule) and reports the fix — now it completes.
628
+ ctl.resolveRound({ round: 3, passed: [], failed: [], undeterminable: [], report: "fixed", findings: [] } as never);
629
+ await settle;
630
+ await __awaitReviewRoundForTests();
631
+ assert.equal(getExecution(), null, "a clean re-report completes");
632
+ ctl.drainAll();
633
+ });
634
+
635
+ it("the two-consecutive-discard synthesis preserves findings instead of clearing them", async () => {
636
+ const { workdir, planPath } = freshWorkdir();
637
+ const ctx = await startTerminal(planPath, workdir);
638
+ const ctl = controlledRunner();
639
+ __setAuditRunnerForTests(ctl.runner);
640
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
641
+ await tick();
642
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [finding("F-001")] } as never);
643
+ await restoring;
644
+ await __awaitReviewRoundForTests();
645
+ const engine = path.join(workdir, "lib", "engine.js");
646
+ const discard = () => {
647
+ fs.writeFileSync(engine, `export const engine = ${Math.random()};\n`, "utf8");
648
+ const later = new Date(Date.now() + 60_000);
649
+ fs.utimesSync(engine, later, later);
650
+ };
651
+ // Re-close, settle, then two fingerprint discards -> the synthesized
652
+ // commit must carry the previous findings forward, not wipe them.
653
+ for (const id of ["Task-1", "Task-2"]) applyTaskUpdate(getExecution()!.tasks, id, "complete", "fixed");
654
+ persistTaskProgress(ctx);
655
+ const settle = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
656
+ await tick();
657
+ discard();
658
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
659
+ await tick();
660
+ discard();
661
+ ctl.resolveRound({ round: 2, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale", findings: [finding("F-001")] } as never);
662
+ await tick();
663
+ await tick();
664
+ const ex = getExecution()!;
665
+ assert.ok(ex, "the synthesis never completes the run");
666
+ assert.equal(ex.audit.rounds, 2, "the synthesis committed as round 2");
667
+ assert.deepEqual(ex.audit.findings.map((f: { id: string }) => f.id), ["F-001"], "findings preserved through the discard synthesis");
668
+ await stopExecution(ctx, "teardown");
669
+ ctl.drainAll();
670
+ await settle;
671
+ await __awaitReviewRoundForTests();
672
+ });
673
+ });
674
+
675
+ describe("plan amendment re-stamps the checkpoint identity (v0.9.1 F-002)", () => {
676
+ it("an amended plan passes /resume-plans instead of plan-mismatch", async () => {
677
+ const { workdir, planPath, runId } = freshWorkdir();
678
+ const ctx = await startTerminal(planPath, workdir);
679
+ const ctl = controlledRunner();
680
+ __setAuditRunnerForTests(ctl.runner);
681
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
682
+ await tick();
683
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-009", severity: "high", taskIds: [], proposedTask: "harden the guard", note: "n", evidence: "e", raw: "r" }] } as never);
684
+ await restoring;
685
+ await __awaitReviewRoundForTests();
686
+ const ex = getExecution()!;
687
+ assert.ok(ex.tasks.find((t) => t.id === "Task-3"), "the amendment appended Task-3");
688
+ // The checkpoint identity now matches the AMENDED file, with provenance.
689
+ const cp = loadCheckpoint(workdir, runId).checkpoint;
690
+ const { sha256File } = await import("../src/workflow-state.ts");
691
+ assert.equal(cp.plan?.sha256, sha256File(planPath), "identity re-stamped to the amended digest");
692
+ assert.equal(cp.execution?.planAmended?.round, 1);
693
+ assert.equal(cp.execution?.planAmended?.sha256, sha256File(planPath));
694
+ // A later resume accepts the amended plan (no plan-mismatch re-approval).
695
+ await stopExecution(ctx, "teardown");
696
+ ctl.drainAll();
697
+ const { loadExecutionFromCheckpoint } = await import("../src/exec.ts");
698
+ const result = loadExecutionFromCheckpoint(makeCtx(workdir), runId);
699
+ assert.equal(result.status, "loaded", `resume accepts the amended plan (${result.status})`);
700
+ assert.deepEqual(result.findings?.map((f) => f.id), ["F-009"], "the load result surfaces unresolved findings for the resume brief (F-005)");
701
+ await stopExecution(makeCtx(workdir), "post-check teardown");
702
+ });
703
+ });
704
+
705
+ describe("executor injection with findings (v0.9)", () => {
706
+ it("executionContextMessage lists unresolved high findings for the repairing agent", async () => {
707
+ const { workdir, planPath } = freshWorkdir();
708
+ const ctx = await startTerminal(planPath, workdir);
709
+ const ctl = controlledRunner();
710
+ __setAuditRunnerForTests(ctl.runner);
711
+ const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
712
+ await tick();
713
+ ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "r1", findings: [{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "loop misses union", evidence: "e", raw: "r" }] } as never);
714
+ await restoring;
715
+ await __awaitReviewRoundForTests();
716
+ const { executionContextMessage } = await import("../src/exec.ts");
717
+ const msg = executionContextMessage(ctx) ?? "";
718
+ assert.match(msg, /unresolved high-severity findings/);
719
+ assert.match(msg, /- F-001 \(Task-2\): loop misses union/);
720
+ assert.match(msg, /Fix them, then re-close the affected tasks/);
721
+ await stopExecution(ctx, "teardown");
722
+ ctl.drainAll();
723
+ });
724
+ });
@@ -331,12 +331,12 @@ describe("refine overlay wiring", () => {
331
331
  it("threads the chrome language through both overlay construction sites (issue #3, VC-007)", () => {
332
332
  const refineSource = fs.readFileSync(path.join(process.cwd(), "tools", "refine.ts"), "utf8");
333
333
  assert.match(refineSource, /const overlayLang = uiLanguageFromTag\(config\.language\.tag\)/);
334
- assert.match(refineSource, /new RefineOverlayController\("reviewer", lanes, relayAbort, lang\)/);
334
+ assert.match(refineSource, /new RefineOverlayController\(\s*"reviewer",\s*lanes,\s*relayAbort,\s*lang,/);
335
335
  assert.match(refineSource, /modelLabel,\s*\n\s*overlayLang,/);
336
336
  const refsSource = fs.readFileSync(path.join(process.cwd(), "tools", "analyze-refs.ts"), "utf8");
337
337
  assert.match(
338
338
  refsSource,
339
- /new RefineOverlayController\("refs", batch\.map\(\(job\) => \(\{ id: job\.laneId, label: job\.laneId \}\)\), relayAbort, resolveUiLanguage\(workdir\)\)/,
339
+ /new RefineOverlayController\(\s*"refs",\s*batch\.map\(\(job\) => \(\{ id: job\.laneId, label: job\.laneId \}\)\),\s*relayAbort,\s*resolveUiLanguage\(workdir\),/,
340
340
  );
341
341
  });
342
342
  });
@@ -533,4 +533,27 @@ describe("refine overlay kitty and fallback key handling (issue #2)", () => {
533
533
  __clearPiTuiForTests();
534
534
  }
535
535
  });
536
+
537
+ it("forwards unhandled keys instead of swallowing them (Ctrl+Shift+T while focused)", () => {
538
+ // pi-tui routes input ONLY to the focused component (no bubbling), so
539
+ // an open overlay used to swallow every global shortcut. Unhandled keys
540
+ // must reach the onUnhandledKey hook so callers can re-dispatch them.
541
+ const calls: string[] = [];
542
+ const component = new RefineOverlayComponent(
543
+ fakeTheme,
544
+ "auditor",
545
+ [readyLane("lane-1", "reviewer-1", "output")],
546
+ () => {},
547
+ undefined,
548
+ undefined,
549
+ "en",
550
+ (data) => calls.push(data),
551
+ );
552
+ // kitty CSI-u form of Ctrl+Shift+T — not an overlay key.
553
+ component.handleInput("\x1b[84;6u");
554
+ assert.deepEqual(calls, ["\x1b[84;6u"], "unhandled keys are forwarded");
555
+ // Handled keys (arrow down) are NOT forwarded.
556
+ component.handleInput("\x1b[B");
557
+ assert.deepEqual(calls, ["\x1b[84;6u"], "handled keys stay internal");
558
+ });
536
559
  });
@@ -482,4 +482,44 @@ describe("blocking budget persistence", () => {
482
482
  const loaded = loadCheckpoint(workdir, runId);
483
483
  assert.equal(loaded.status, "corrupt", "the closed schema is still enforced");
484
484
  });
485
+ // v0.9: findings ride the same audit slot — same persistence rules.
486
+
487
+ it("round-trips findings through the checkpoint validator", () => {
488
+ const { workdir, runId } = setupRun("findings-roundtrip");
489
+ const cp = applyExecutionProgress(executingCheckpoint(workdir, runId), {
490
+ audit: {
491
+ rounds: 2,
492
+ lastResult: "highs: F-001",
493
+ findings: [
494
+ { id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "src/b.ts", raw: "raw line" },
495
+ { id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy", note: "polish", evidence: "e", raw: "r" },
496
+ ],
497
+ },
498
+ });
499
+ mutateCheckpoint(workdir, runId, () => cp);
500
+ const loaded = loadCheckpoint(workdir, runId);
501
+ assert.ok(loaded.status === "ok");
502
+ assert.deepEqual(loaded.checkpoint.execution?.audit?.findings, [
503
+ { id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "src/b.ts", raw: "raw line" },
504
+ { id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy", note: "polish", evidence: "e", raw: "r" },
505
+ ]);
506
+ });
507
+
508
+ it("loads a checkpoint written before findings existed (absent, not corrupt)", () => {
509
+ const { workdir, runId } = setupRun("findings-legacy");
510
+ const cp = applyExecutionProgress(executingCheckpoint(workdir, runId), { audit: { rounds: 1, lastResult: "VC-001" } });
511
+ mutateCheckpoint(workdir, runId, () => cp);
512
+ const loaded = loadCheckpoint(workdir, runId);
513
+ assert.ok(loaded.status === "ok", "legacy checkpoint still loads");
514
+ assert.equal(loaded.checkpoint.execution?.audit?.findings, undefined);
515
+ });
516
+
517
+ it("rejects unknown keys inside a findings record", () => {
518
+ const { workdir, runId } = setupRun("findings-bad-record");
519
+ const cp = applyExecutionProgress(executingCheckpoint(workdir, runId), {
520
+ audit: { rounds: 1, findings: [{ id: "F-001", severity: "high", taskIds: [], note: "n", evidence: "e", raw: "r", bogus: true } as never] },
521
+ });
522
+ // Validation runs at write: a malformed findings record never lands.
523
+ assert.throws(() => mutateCheckpoint(workdir, runId, () => cp), /findings\.0: unexpected key/);
524
+ });
485
525
  });
@@ -15,7 +15,6 @@
15
15
 
16
16
  import { StringEnum } from "@earendil-works/pi-ai";
17
17
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
18
- import { truncateHead } from "@earendil-works/pi-coding-agent";
19
18
  import { Text } from "@earendil-works/pi-tui";
20
19
  import { Type } from "typebox";
21
20
  import * as fs from "node:fs";
@@ -37,6 +36,8 @@ import { resolveActiveRun } from "../src/run-context.ts";
37
36
  import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
38
37
  import { runPiSubagent, stripFrontmatter } from "../src/subagent.ts";
39
38
  import { RefineOverlayController, refineOverlayContext } from "../src/refine-ui.ts";
39
+ import { matchesTerminalKey } from "../src/terminal-keys.ts";
40
+ import { toggleDashboardExpanded } from "../src/exec.ts";
40
41
  import { resolveUiLanguage } from "../src/ui-language.ts";
41
42
 
42
43
  const BATCH_SIZE = 3;
@@ -80,7 +81,7 @@ async function ensureRefAnalystModelReady(
80
81
  ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
81
82
  if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
82
83
  let outcome = await runFirstUseFlow(host, role.thinking_level);
83
- if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
84
+ if (outcome.status === "confirmed" && availableModels(host).length > 0 && findModel(host, outcome.modelSelector) === null) {
84
85
  // F-008: a manually entered selector that the registry does not know —
85
86
  // one re-pick, then let spawn-side errors surface precisely.
86
87
  outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
@@ -226,7 +227,16 @@ export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): voi
226
227
 
227
228
  const overlay =
228
229
  ctx.mode === "tui"
229
- ? new RefineOverlayController("refs", batch.map((job) => ({ id: job.laneId, label: job.laneId })), relayAbort, resolveUiLanguage(workdir))
230
+ ? new RefineOverlayController(
231
+ "refs",
232
+ batch.map((job) => ({ id: job.laneId, label: job.laneId })),
233
+ relayAbort,
234
+ resolveUiLanguage(workdir),
235
+ // Forward the dashboard toggle (no key bubbling in pi-tui).
236
+ (data) => {
237
+ if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
238
+ },
239
+ )
230
240
  : undefined;
231
241
  overlay?.open(refineOverlayContext(ctx), modelLabel);
232
242
  try {
@@ -264,9 +274,10 @@ export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): voi
264
274
  }
265
275
 
266
276
  const combined = sections.join("\n\n---\n\n");
267
- const truncation = truncateHead(combined, { maxLines: 2000, maxBytes: 50 * 1024 });
268
- let text = truncation.content;
269
- if (truncation.truncated) text += `\n\n[Output truncated; full outputs remain in this tool result's details.]`;
277
+ // v0.8.1: no head-truncation — the full combined analysis flows into the
278
+ // tool result (and REF_ANALYSIS.md) verbatim; the previous 2000-line/
279
+ // 50KB cap silently dropped the tail of large reference analyses.
280
+ const text = combined;
270
281
 
271
282
  return {
272
283
  content: [
package/tools/refine.ts CHANGED
@@ -44,6 +44,8 @@ import { buildReviewerTask, reviewerLanes } from "../src/refine-prompts.ts";
44
44
  import { graphBlockForRefiner } from "../src/code-graph/prompts.ts";
45
45
  import { runPiSubagent, stripFrontmatter } from "../src/subagent.ts";
46
46
  import { RefineOverlayController, refineOverlayContext } from "../src/refine-ui.ts";
47
+ import { matchesTerminalKey } from "../src/terminal-keys.ts";
48
+ import { toggleDashboardExpanded } from "../src/exec.ts";
47
49
 
48
50
 
49
51
  const RefineParams = Type.Object({
@@ -90,10 +92,15 @@ async function ensureReviewerReady(
90
92
  ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
91
93
  if (role.mode === "current-session" || reviewerReady(role as never)) return role as never;
92
94
  let outcome: FirstUseOutcome = await runFirstUseFlow(host, role.thinking_level);
93
- if (outcome.status === "confirmed" && outcome.model_selector !== null) {
95
+ if (outcome.status === "confirmed") {
94
96
  // F-008: validate the freshly chosen selector against the registry when
95
97
  // one is present, so a typo'd manual entry fails here, not at spawn.
96
- if (availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
98
+ // (v0.8.1 field-drift fix: the outcome's top-level field is camelCase
99
+ // `modelSelector` — reading snake_case `model_selector` yielded
100
+ // undefined, passed the old `!== null` guard, and crashed findModel
101
+ // with "Cannot read properties of undefined (reading 'indexOf')"
102
+ // right after the first-use panel confirmed.)
103
+ if (availableModels(host).length > 0 && findModel(host, outcome.modelSelector) === null) {
97
104
  outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
98
105
  }
99
106
  }
@@ -117,7 +124,19 @@ function setupRefinementExecution(
117
124
  if (parentSignal?.aborted) controller.abort();
118
125
  else parentSignal?.addEventListener("abort", relayAbort, { once: true });
119
126
 
120
- const overlay = ctx.mode === "tui" ? new RefineOverlayController("reviewer", lanes, relayAbort, lang) : undefined;
127
+ const overlay = ctx.mode === "tui"
128
+ ? new RefineOverlayController(
129
+ "reviewer",
130
+ lanes,
131
+ relayAbort,
132
+ lang,
133
+ // pi-tui has no key bubbling: forward the dashboard toggle so
134
+ // Ctrl+Shift+T keeps working while the refine overlay holds focus.
135
+ (data) => {
136
+ if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
137
+ },
138
+ )
139
+ : undefined;
121
140
  overlay?.open(refineOverlayContext(ctx), modelLabel);
122
141
 
123
142
  return {