pi-plans 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +5 -12
  3. package/README.md +5 -5
  4. package/agents/execution-reviewer.md +92 -0
  5. package/index.ts +13 -23
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +14 -7
  8. package/references/plan-artifact-template.md +11 -1
  9. package/references/state-and-config.md +3 -3
  10. package/scripts/validate.ts +20 -3
  11. package/src/auditor.ts +306 -63
  12. package/src/code-graph/commands.ts +6 -1
  13. package/src/dashboard.ts +91 -13
  14. package/src/exec.ts +835 -142
  15. package/src/plan.ts +1 -1
  16. package/src/refine-ui-state.ts +1 -1
  17. package/src/refine-ui.ts +40 -8
  18. package/src/resume-command.ts +19 -3
  19. package/src/resume.ts +5 -1
  20. package/src/staleness.ts +53 -0
  21. package/src/state.ts +1 -0
  22. package/src/task-tool.ts +1 -1
  23. package/src/tasks.ts +62 -5
  24. package/src/ui-language.ts +4 -0
  25. package/src/workflow-state.ts +93 -6
  26. package/tests/analyze-refs.test.ts +1 -1
  27. package/tests/auditor.test.ts +299 -16
  28. package/tests/dashboard.test.ts +202 -2
  29. package/tests/exec-review-loop.test.ts +724 -0
  30. package/tests/exec.test.ts +198 -44
  31. package/tests/extension-load.test.ts +1 -1
  32. package/tests/refine-ui.test.ts +25 -2
  33. package/tests/resume-lifecycle.test.ts +5 -1
  34. package/tests/resume.test.ts +6 -0
  35. package/tests/staleness.test.ts +76 -0
  36. package/tests/state.test.ts +4 -0
  37. package/tests/tasks.test.ts +142 -0
  38. package/tests/workflow-state.test.ts +105 -0
  39. package/tools/analyze-refs.ts +17 -6
  40. package/tools/execute-plan.ts +12 -5
  41. package/tools/plans.ts +1 -1
  42. package/tools/refine.ts +22 -3
@@ -11,6 +11,7 @@ import * as os from "node:os";
11
11
  import * as path from "node:path";
12
12
  import { after, before, describe, it } from "node:test";
13
13
  import {
14
+ __awaitReviewRoundForTests,
14
15
  __setAuditRunnerForTests,
15
16
  executionContextMessage,
16
17
  getExecution,
@@ -23,6 +24,7 @@ import {
23
24
  toggleDashboardExpanded,
24
25
  updateStatusWidget,
25
26
  } from "../src/exec.ts";
27
+ import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
26
28
  import { setMessagingApi } from "../src/messaging.ts";
27
29
  import { applyTaskUpdate } from "../src/task-tool.ts";
28
30
  import { flattenTaskViews } from "../src/tasks.ts";
@@ -157,7 +159,6 @@ describe("task-tree execution core", () => {
157
159
  return item.id;
158
160
  }),
159
161
  failed: [],
160
- rolledBack: [],
161
162
  report: "all pass",
162
163
  }));
163
164
  const { workdir, planPath, runId } = freshWorkdir();
@@ -186,7 +187,7 @@ describe("task-tree execution core", () => {
186
187
  round += 1;
187
188
  const failed = checklist.filter((item) => item.id === "VC-002").map((item) => item.id);
188
189
  // Simulate the pure outcome: VC-002 fails; its covered tasks roll back.
189
- return { round, passed: ["VC-001"], failed, rolledBack: [], report: "VC-002 fails" };
190
+ return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
190
191
  });
191
192
  const { workdir, planPath, runId } = freshWorkdir();
192
193
  const { ctx } = await start(planPath, workdir);
@@ -300,7 +301,7 @@ describe("task-tree execution core", () => {
300
301
  // Mirror applyAuditOutcome: passed checks are marked done.
301
302
  const vc1 = checklist.find((item) => item.id === "VC-001");
302
303
  if (vc1) vc1.done = true;
303
- return { round: 1, passed: ["VC-001"], failed: ["VC-002"], rolledBack: [], report: "VC-002 fails: core tests missing" };
304
+ return { round: 1, passed: ["VC-001"], failed: ["VC-002"], report: "VC-002 fails: core tests missing" };
304
305
  });
305
306
  const { workdir, planPath, runId } = freshWorkdir();
306
307
  const { ctx } = await start(planPath, workdir);
@@ -323,10 +324,10 @@ describe("task-tree execution core", () => {
323
324
  await stopExecution(ctx, "test teardown");
324
325
  });
325
326
 
326
- it("audit round cap: interactive sessions pause, headless stops", async () => {
327
+ it("review round cap: every mode pauses with an in-band signal (v0.8)", async () => {
327
328
  __setAuditRunnerForTests(async ({ checklist }) => {
328
329
  const failed = checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id);
329
- return { round: 1, passed: ["VC-001"], failed, rolledBack: [], report: "still failing" };
330
+ return { round: 1, passed: ["VC-001"], failed, report: "still failing" };
330
331
  });
331
332
  const interactive = await (async () => {
332
333
  const { workdir, planPath, runId } = freshWorkdir();
@@ -335,36 +336,43 @@ describe("task-tree execution core", () => {
335
336
  applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
336
337
  persistTaskProgress(ctx);
337
338
  }
338
- // Simulate three exhausted rounds persisted from earlier attempts.
339
- getExecution()!.audit.rounds = 3;
339
+ // Simulate an exhausted budget persisted from earlier attempts.
340
+ getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
340
341
  getExecution()!.audit.failed = ["VC-002"];
341
- mutateCheckpoint(workdir, runId!, (cp) => applyExecutionProgress(cp, { audit: { rounds: 3, lastResult: "VC-002" } }));
342
+ mutateCheckpoint(workdir, runId!, (cp) => applyExecutionProgress(cp, { audit: { rounds: REVIEW_MAX_ROUNDS, lastResult: "VC-002" } }));
342
343
  const snapshot = getExecution();
343
344
  const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
344
345
  await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
346
+ await __awaitReviewRoundForTests(); // detached chain: pause lands inside it
345
347
  const ex = getExecution()!;
346
- // hasUI ctx → interactive pause, execution state kept.
348
+ // v0.8 (Q-A): interactive AND headless both pause — fail-closed, never a silent stop.
347
349
  assert.equal(ex.stall.paused, true, "interactive cap pauses");
348
- assert.match(ex.stall.pausedReason ?? "", /exhausted 3 rounds/);
350
+ assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
351
+ assert.ok(
352
+ ctxUi.entries.some((e) => e.customType === "pi-plans-review-paused"),
353
+ "the pause lands as an in-band message every mode can read",
354
+ );
349
355
  await stopExecution(ctxUi, "test teardown");
350
356
  return loadCheckpoint(workdir, runId!);
351
357
  })();
352
358
  assert.ok(interactive.status === "ok");
353
- // Headless: no UI → bounded stop.
359
+ // Headless: same pause, execution state kept (no more bounded stop).
354
360
  const { workdir: wd2, planPath: pp2, runId: r2 } = freshWorkdir();
355
361
  const { ctx: ctx3 } = await start(pp2, wd2);
356
362
  for (const id of ["Task-1", "Task-2", "Task-3"]) {
357
363
  applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
358
364
  persistTaskProgress(ctx3);
359
365
  }
360
- getExecution()!.audit.rounds = 3;
366
+ getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
361
367
  getExecution()!.audit.failed = ["VC-002"];
362
368
  const headlessCtx = { ...makeCtx(wd2), hasUI: false } as never;
363
369
  await restoreFromSession(headlessCtx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
364
- assert.equal(getExecution(), null, "headless cap stops and clears execution");
365
- const stopped = loadCheckpoint(wd2, r2!);
366
- assert.ok(stopped.status === "ok");
367
- assert.match(stopped.checkpoint.execution?.pausedReason ?? "", /exhausted 3 rounds/);
370
+ assert.notEqual(getExecution(), null, "headless cap pauses and keeps execution state");
371
+ assert.equal(getExecution()!.stall.paused, true, "headless pauses too (v0.8)");
372
+ const paused = loadCheckpoint(wd2, r2!);
373
+ assert.ok(paused.status === "ok");
374
+ assert.match(paused.checkpoint.execution?.pausedReason ?? "", /execution review exhausted 5 rounds/);
375
+ await stopExecution(makeCtx(wd2), "test teardown");
368
376
  __setAuditRunnerForTests(null);
369
377
  });
370
378
 
@@ -383,9 +391,10 @@ describe("task-tree execution core", () => {
383
391
  assert.equal(getExecution(), null, "delegate orphans never resume without a fresh handoff");
384
392
  });
385
393
 
386
- it("resuming an audit-cap pause grants a fresh audit budget and completes", async () => {
387
- // Round 2 F-001 regression nail: cap → pause → resume resets rounds →
388
- // the audit runs again and can now complete the run.
394
+ it("resuming a review-cap pause via /plans-execute grants a fresh budget and completes (v0.8)", async () => {
395
+ // Round 2 F-001 regression nail, v0.8 form: cap → pause → the explicit
396
+ // /plans-execute surface (resumeActiveExecution) resets rounds → the
397
+ // review runs again and can now complete the run. Ordinary input must NOT.
389
398
  let runnerCalls = 0;
390
399
  __setAuditRunnerForTests(async ({ checklist }) => {
391
400
  runnerCalls += 1;
@@ -396,7 +405,6 @@ describe("task-tree execution core", () => {
396
405
  return item.id;
397
406
  }),
398
407
  failed: [],
399
- rolledBack: [],
400
408
  report: "all pass",
401
409
  };
402
410
  });
@@ -406,30 +414,40 @@ describe("task-tree execution core", () => {
406
414
  applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
407
415
  persistTaskProgress(ctx);
408
416
  }
409
- // Exhaust the budget, then pause at the cap exactly like runAuditFlow.
410
- getExecution()!.audit.rounds = 3;
417
+ // Exhaust the budget, then pause at the cap exactly like the loop does.
418
+ getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
411
419
  getExecution()!.audit.failed = ["VC-001", "VC-002"];
412
420
  const ctxTui = { ...makeCtx(workdir), mode: "tui" as const };
413
421
  const snapshot = getExecution();
414
422
  await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
415
- const ex = getExecution()!;
416
- assert.equal(ex.stall.paused, true, "cap pauses interactively");
417
- assert.match(ex.stall.pausedReason ?? "", /^completion audit exhausted/);
418
- // User resumes: the budget resets and the next audit flow completes.
423
+ await __awaitReviewRoundForTests();
424
+ let ex = getExecution()!;
425
+ assert.equal(ex.stall.paused, true, "cap pauses");
426
+ assert.match(ex.stall.pausedReason ?? "", /^execution review exhausted/);
427
+ // Ordinary input does NOT lift a review-cap pause (CF2-004).
428
+ const beforeRounds = ex.audit.rounds;
429
+ const { resumeGoalWaitIfPaused } = await import("../src/exec.ts");
430
+ const viaInput = resumeGoalWaitIfPaused(ctxTui);
431
+ assert.equal(viaInput, false, "input never lifts a review-cap pause");
432
+ assert.equal(getExecution()!.stall.paused, true, "still paused after input");
433
+ assert.equal(getExecution()!.audit.rounds, beforeRounds, "budget survives input");
434
+ // The explicit surface grants the fresh budget and completes the run.
419
435
  const { resumeActiveExecution } = await import("../src/exec.ts");
420
436
  const resumed = resumeActiveExecution(ctxTui);
421
437
  assert.equal(resumed, true);
422
- assert.equal(getExecution()!.audit.rounds, 0, "resume grants a fresh audit budget");
423
- const snapshot2 = getExecution();
424
- await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: snapshot2 }]);
425
- assert.equal(runnerCalls, 1, "audit re-ran after the resume (fresh budget)");
438
+ assert.equal(getExecution()!.audit.rounds, 0, "resume grants a fresh review budget");
439
+ await __awaitReviewRoundForTests(); // detached grant chain completes the run
440
+ assert.equal(runnerCalls, 1, "review re-ran after the resume (fresh budget)");
426
441
  const final = loadCheckpoint(workdir, runId!);
427
442
  assert.ok(final.status === "ok");
428
443
  assert.equal(final.checkpoint.phase, "completed");
429
444
  __setAuditRunnerForTests(null);
430
445
  });
431
446
 
432
- it("a null audit outcome (infra failure) fails every pending check closed", async () => {
447
+ it("a null review outcome stays undeterminable, self-schedules to the cap, and pauses (v0.8)", async () => {
448
+ // An infra failure is not evidence of bad work: it must not complete the
449
+ // run, roll work back, or accuse the agent — and the loop retries it
450
+ // internally until the budget is spent, then pauses fail-closed.
433
451
  __setAuditRunnerForTests(async () => null);
434
452
  const { workdir, planPath, runId } = freshWorkdir();
435
453
  const { ctx } = await start(planPath, workdir);
@@ -440,25 +458,36 @@ describe("task-tree execution core", () => {
440
458
  const snapshot = getExecution();
441
459
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
442
460
  const ex = getExecution()!;
443
- assert.deepEqual(ex.audit.failed, ["VC-001", "VC-002"], "unreported checks fail closed");
461
+ assert.deepEqual(ex.audit.undeterminable, ["VC-001", "VC-002"], "unreported checks are undeterminable");
462
+ assert.deepEqual(ex.audit.failed, [], "nothing was judged wrong");
444
463
  for (const id of ["Task-1", "Task-2", "Task-3"]) {
445
- assert.equal(ex.tasks.find((t) => t.id === id)?.status, "pending", `${id} rolled back`);
464
+ assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
446
465
  }
466
+ assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "the retry chain spent the whole budget");
467
+ assert.equal(ex.stall.paused, true, "the loop pauses at the cap");
468
+ assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
469
+ assert.equal(
470
+ ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable").length,
471
+ 0,
472
+ "undeterminable rounds never wake the agent (self-scheduled retries)",
473
+ );
474
+ assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the pause is in-band");
447
475
  const load = loadCheckpoint(workdir, runId!);
448
476
  assert.ok(load.status === "ok");
449
- assert.equal(load.checkpoint.execution?.audit?.rounds, 1);
450
- assert.match(load.checkpoint.execution?.audit?.lastResult ?? "", /VC-001/);
477
+ assert.equal(load.checkpoint.phase, "executing", "an unreadable round must not complete the run");
478
+ assert.equal(load.checkpoint.execution?.audit?.rounds, REVIEW_MAX_ROUNDS);
451
479
  __setAuditRunnerForTests(null);
452
480
  await stopExecution(ctx, "test teardown");
453
481
  });
454
482
 
455
- it("a runner passing a strict subset fails the unreported checks closed", async () => {
483
+ it("a runner passing a strict subset credits the pass and keeps the rest undeterminable", async () => {
456
484
  // The runner reports VC-001 passed and claims zero failures — VC-002
457
- // is simply missing from its report and must NOT complete the run.
485
+ // is simply missing from its report, so it must NOT complete the run,
486
+ // must NOT be called failed, and must NOT roll Task-3 back.
458
487
  __setAuditRunnerForTests(async ({ checklist }) => {
459
488
  const vc1 = checklist.find((item) => item.id === "VC-001");
460
489
  if (vc1) vc1.done = true;
461
- return { round: 1, passed: ["VC-001"], failed: [], rolledBack: [], report: "partial report" };
490
+ return { round: 1, passed: ["VC-001"], failed: [], undeterminable: ["VC-002"], report: "partial report" };
462
491
  });
463
492
  const { workdir, planPath } = freshWorkdir();
464
493
  const { ctx } = await start(planPath, workdir);
@@ -469,12 +498,137 @@ describe("task-tree execution core", () => {
469
498
  const snapshot = getExecution();
470
499
  await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
471
500
  const ex = getExecution()!;
472
- assert.deepEqual(ex.audit.failed, ["VC-002"], "unreported check fails despite failed: []");
473
- assert.equal(ex.tasks.find((t) => t.id === "Task-3")?.status, "pending");
501
+ assert.deepEqual(ex.audit.undeterminable, ["VC-002"], "the unreported check stays pending judgement");
502
+ assert.deepEqual(ex.audit.failed, []);
503
+ assert.equal(ex.tasks.find((t) => t.id === "Task-3")?.status, "complete", "partial pass does not roll back");
504
+ __setAuditRunnerForTests(null);
505
+ await stopExecution(ctx, "test teardown");
506
+ });
507
+
508
+ it("an emphasized pass verdict completes the run with no rollback", async () => {
509
+ // The production incident: the auditor wrote `verdict: **pass**` for
510
+ // every check and the run rolled everything back. Emphasis must parse,
511
+ // and a fully-passing round must complete rather than pause.
512
+ __setAuditRunnerForTests(async ({ checklist }) => ({
513
+ round: 1,
514
+ passed: checklist.map((item) => item.id),
515
+ failed: [],
516
+ undeterminable: [],
517
+ report: "- `VC-001` — verdict: **pass**\n- `VC-002` — verdict: **pass**",
518
+ }));
519
+ const { workdir, planPath, runId } = freshWorkdir();
520
+ const { ctx } = await start(planPath, workdir);
521
+ for (const id of ["Task-1", "Task-2", "Task-3"]) {
522
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
523
+ persistTaskProgress(ctx);
524
+ }
525
+ const snapshot = getExecution();
526
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
527
+ const load = loadCheckpoint(workdir, runId!);
528
+ assert.ok(load.status === "ok");
529
+ assert.equal(load.checkpoint.phase, "completed", "a fully-passing round completes the run");
530
+ assert.equal(load.checkpoint.execution?.audit?.passed, true);
531
+ __setAuditRunnerForTests(null);
532
+ await stopExecution(makeCtx(workdir), "test teardown");
533
+ });
534
+
535
+ it("an all-undeterminable round completes nothing, rolls nothing back, and never wakes (v0.8)", async () => {
536
+ // Fail-open guard: `undeterminable` is neither a pass nor a failure, so
537
+ // an all-undeterminable round must leave failed === [] AND must not take
538
+ // the completion branch. v0.8: retries self-schedule (Q-B) — zero wakes.
539
+ __setAuditRunnerForTests(async ({ checklist }) => ({
540
+ round: 1,
541
+ passed: [],
542
+ failed: [],
543
+ undeterminable: checklist.map((item) => item.id),
544
+ report: "## Findings\n\n- `F-001` — severity: high; verdict-free reviewer output\n\n## Questions\n\nNone.",
545
+ }));
546
+ const { workdir, planPath, runId } = freshWorkdir();
547
+ const { ctx } = await start(planPath, workdir);
548
+ for (const id of ["Task-1", "Task-2", "Task-3"]) {
549
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
550
+ persistTaskProgress(ctx);
551
+ }
552
+ const snapshot = getExecution();
553
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: snapshot }]);
554
+ const ex = getExecution()!;
555
+ assert.deepEqual(ex.audit.failed, []);
556
+ assert.deepEqual(ex.audit.undeterminable.sort(), ["VC-001", "VC-002"]);
557
+ for (const id of ["Task-1", "Task-2", "Task-3"]) {
558
+ assert.equal(ex.tasks.find((t) => t.id === id)?.status, "complete", `${id} not rolled back`);
559
+ }
560
+ assert.equal(ex.audit.rounds, REVIEW_MAX_ROUNDS, "self-scheduled retries spent the budget");
561
+ assert.equal(ex.stall.paused, true, "paused at the cap");
562
+ const load = loadCheckpoint(workdir, runId!);
563
+ assert.ok(load.status === "ok");
564
+ assert.equal(load.checkpoint.phase, "executing", "never fail-open into completed");
565
+ // Undeterminable rounds never wake the agent (Q-B); only the pause is announced.
566
+ const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-undeterminable");
567
+ assert.equal(wakes.length, 0, "undeterminable rounds send no wake");
568
+ assert.ok(ctx.entries.some((e) => e.customType === "pi-plans-review-paused"), "the cap pause is in-band");
474
569
  __setAuditRunnerForTests(null);
475
570
  await stopExecution(ctx, "test teardown");
476
571
  });
477
572
 
573
+ it("the audit budget is monotonic: an agent repair re-close does not refund a round", async () => {
574
+ // The unbounded-loop guard. The agent's repair is a forward transition
575
+ // (pending -> complete); if that reset the audit budget, the sequence
576
+ // fail -> repair -> fail would never reach the cap.
577
+ let call = 0;
578
+ __setAuditRunnerForTests(async ({ checklist, round }) => {
579
+ call += 1;
580
+ return call === 1
581
+ ? { round, passed: ["VC-001"], failed: ["VC-002"], report: "first attempt fails" }
582
+ : { round, passed: ["VC-001"], failed: ["VC-002"], report: "still failing" };
583
+ });
584
+ const { workdir, planPath } = freshWorkdir();
585
+ const { ctx } = await start(planPath, workdir);
586
+ for (const id of ["Task-1", "Task-2", "Task-3"]) {
587
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
588
+ persistTaskProgress(ctx);
589
+ }
590
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
591
+ assert.equal(getExecution()!.audit.rounds, 1);
592
+ // The agent repairs the rolled-back work and re-closes it: a forward
593
+ // transition, exactly what a naive budget reset would reward.
594
+ for (const id of ["Task-2", "Task-3"]) {
595
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence again`);
596
+ persistTaskProgress(ctx);
597
+ }
598
+ await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
599
+ assert.equal(getExecution()!.audit.rounds, 2, "the second attempt costs a second round");
600
+ __setAuditRunnerForTests(null);
601
+ await stopExecution(makeCtx(workdir), "test teardown");
602
+ });
603
+
604
+ it("the audit round cap test still pauses an interactive session", async () => {
605
+ // Unchanged intent: exhausting the budget pauses interactively. The
606
+ // seam moved (getExecution()!.audit.rounds), so it is pinned here.
607
+ __setAuditRunnerForTests(async ({ checklist }) => ({
608
+ round: 1,
609
+ passed: [],
610
+ failed: checklist.filter((item) => !["VC-001"].includes(item.id)).map((item) => item.id),
611
+ undeterminable: [],
612
+ report: "still failing",
613
+ }));
614
+ const { workdir, planPath } = freshWorkdir();
615
+ const { ctx } = await start(planPath, workdir);
616
+ for (const id of ["Task-1", "Task-2", "Task-3"]) {
617
+ applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
618
+ persistTaskProgress(ctx);
619
+ }
620
+ getExecution()!.audit.rounds = REVIEW_MAX_ROUNDS;
621
+ getExecution()!.audit.failed = ["VC-002"];
622
+ const ctxUi = { ...makeCtx(workdir), mode: "tui" as const };
623
+ await restoreFromSession(ctxUi, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
624
+ await __awaitReviewRoundForTests();
625
+ const ex = getExecution()!;
626
+ assert.equal(ex.stall.paused, true, "the review cap pauses an interactive session");
627
+ assert.match(ex.stall.pausedReason ?? "", /execution review exhausted 5 rounds/);
628
+ __setAuditRunnerForTests(null);
629
+ await stopExecution(ctxUi, "test teardown");
630
+ });
631
+
478
632
  it("toggleDashboardExpanded flips the expanded mode", () => {
479
633
  const before = toggleEnabled();
480
634
  toggleDashboardExpanded(makeCtx(freshWorkdir(false).workdir));
@@ -557,10 +711,10 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
557
711
  round += 1;
558
712
  if (round === 1) {
559
713
  const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
560
- return { round, passed: ["VC-001"], failed, rolledBack: [], report: "VC-002 fails" };
714
+ return { round, passed: ["VC-001"], failed, report: "VC-002 fails" };
561
715
  }
562
716
  for (const item of checklist) item.done = true;
563
- return { round, passed: checklist.map((i) => i.id), failed: [], rolledBack: [], report: "all pass" };
717
+ return { round, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
564
718
  });
565
719
  const { workdir, planPath, runId } = freshWorkdir();
566
720
  const ctx = await startTui(planPath, workdir);
@@ -598,7 +752,7 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
598
752
  __setAuditRunnerForTests(async ({ checklist }) => {
599
753
  calls += 1;
600
754
  for (const item of checklist) item.done = true;
601
- return { round: 1, passed: checklist.map((i) => i.id), failed: [], rolledBack: [], report: "all pass" };
755
+ return { round: 1, passed: checklist.map((i) => i.id), failed: [], report: "all pass" };
602
756
  });
603
757
  const { workdir, planPath } = freshWorkdir();
604
758
  const ctx = await startTui(planPath, workdir);
@@ -630,7 +784,7 @@ describe("v0.7.1 execution-loop fixes (event-driven)", () => {
630
784
  it("root cause A: an audit failure wakes the agent exactly once", async () => {
631
785
  __setAuditRunnerForTests(async ({ checklist }) => {
632
786
  const failed = checklist.filter((i) => i.id === "VC-002").map((i) => i.id);
633
- return { round: 1, passed: ["VC-001"], failed, rolledBack: [], report: "VC-002 fails" };
787
+ return { round: 1, passed: ["VC-001"], failed, report: "VC-002 fails" };
634
788
  });
635
789
  const { workdir, planPath } = freshWorkdir();
636
790
  const ctx = await startTui(planPath, workdir);
@@ -4,7 +4,7 @@
4
4
  * WITHOUT typechecking — duplicate type/interface declarations pass silently.
5
5
  * pi's extension loader parses TypeScript for real and refused the whole
6
6
  * extension ("Identifier 'RoleConfig' has already been declared"), which made
7
- * every spawned subagent (including the completion auditor) fail while the
7
+ * every spawned subagent (including the execution reviewer) fail while the
8
8
  * whole suite stayed green.
9
9
  *
10
10
  * Guard: spawn `pi` from the repo root with a deliberately missing model.
@@ -331,12 +331,12 @@ describe("refine overlay wiring", () => {
331
331
  it("threads the chrome language through both overlay construction sites (issue #3, VC-007)", () => {
332
332
  const refineSource = fs.readFileSync(path.join(process.cwd(), "tools", "refine.ts"), "utf8");
333
333
  assert.match(refineSource, /const overlayLang = uiLanguageFromTag\(config\.language\.tag\)/);
334
- assert.match(refineSource, /new RefineOverlayController\("reviewer", lanes, relayAbort, lang\)/);
334
+ assert.match(refineSource, /new RefineOverlayController\(\s*"reviewer",\s*lanes,\s*relayAbort,\s*lang,/);
335
335
  assert.match(refineSource, /modelLabel,\s*\n\s*overlayLang,/);
336
336
  const refsSource = fs.readFileSync(path.join(process.cwd(), "tools", "analyze-refs.ts"), "utf8");
337
337
  assert.match(
338
338
  refsSource,
339
- /new RefineOverlayController\("refs", batch\.map\(\(job\) => \(\{ id: job\.laneId, label: job\.laneId \}\)\), relayAbort, resolveUiLanguage\(workdir\)\)/,
339
+ /new RefineOverlayController\(\s*"refs",\s*batch\.map\(\(job\) => \(\{ id: job\.laneId, label: job\.laneId \}\)\),\s*relayAbort,\s*resolveUiLanguage\(workdir\),/,
340
340
  );
341
341
  });
342
342
  });
@@ -533,4 +533,27 @@ describe("refine overlay kitty and fallback key handling (issue #2)", () => {
533
533
  __clearPiTuiForTests();
534
534
  }
535
535
  });
536
+
537
+ it("forwards unhandled keys instead of swallowing them (Ctrl+Shift+T while focused)", () => {
538
+ // pi-tui routes input ONLY to the focused component (no bubbling), so
539
+ // an open overlay used to swallow every global shortcut. Unhandled keys
540
+ // must reach the onUnhandledKey hook so callers can re-dispatch them.
541
+ const calls: string[] = [];
542
+ const component = new RefineOverlayComponent(
543
+ fakeTheme,
544
+ "auditor",
545
+ [readyLane("lane-1", "reviewer-1", "output")],
546
+ () => {},
547
+ undefined,
548
+ undefined,
549
+ "en",
550
+ (data) => calls.push(data),
551
+ );
552
+ // kitty CSI-u form of Ctrl+Shift+T — not an overlay key.
553
+ component.handleInput("\x1b[84;6u");
554
+ assert.deepEqual(calls, ["\x1b[84;6u"], "unhandled keys are forwarded");
555
+ // Handled keys (arrow down) are NOT forwarded.
556
+ component.handleInput("\x1b[B");
557
+ assert.deepEqual(calls, ["\x1b[84;6u"], "handled keys stay internal");
558
+ });
536
559
  });
@@ -259,7 +259,11 @@ describe("/resume-plans on the real Pi host", () => {
259
259
  assert.equal(final.status, "ok");
260
260
  if (final.status === "ok") {
261
261
  assert.equal(final.checkpoint.execution?.tasks?.["Task-1"]?.status, "complete");
262
- assert.equal(final.checkpoint.execution?.tasks?.["Task-2"]?.status, undefined);
262
+ // Every task now carries a record: an untouched task is
263
+ // persisted as pending rather than omitted, so the checkpoint
264
+ // distinguishes "never started" from "rolled back with
265
+ // evidence" instead of losing the record entirely.
266
+ assert.equal(final.checkpoint.execution?.tasks?.["Task-2"]?.status, "pending");
263
267
  }
264
268
  } finally {
265
269
  session2.session.dispose();
@@ -112,6 +112,9 @@ describe("candidate discovery", () => {
112
112
  it("lists resumable runs, excludes terminal ones, registry hint first", () => {
113
113
  const workdir = setupRepo("discovery");
114
114
  const planning = startRun(workdir, { topic: "alpha", skill: "plan-normal", requestText: "a" }).run;
115
+ const verifying = startRun(workdir, { topic: "zeta", skill: "plan-normal", requestText: "f" }).run;
116
+ setRunStatus(workdir, verifying.run_id, "executing");
117
+ setRunStatus(workdir, verifying.run_id, "verifying");
115
118
  const stopped = startRun(workdir, { topic: "beta", skill: "plan-normal", requestText: "b" }).run;
116
119
  setRunStatus(workdir, stopped.run_id, "executing");
117
120
  setRunStatus(workdir, stopped.run_id, "stopped");
@@ -129,6 +132,9 @@ describe("candidate discovery", () => {
129
132
  assert.ok(ids.includes(planning.run_id));
130
133
  assert.ok(ids.includes(stopped.run_id));
131
134
  assert.ok(ids.includes(doneWithReview.run_id), "done with review artifacts resumable");
135
+ assert.ok(ids.includes(verifying.run_id), "verifying runs are resumable");
136
+ const verifyingCandidate = candidates.find((candidate) => candidate.runId === verifying.run_id);
137
+ assert.equal(verifyingCandidate?.phaseLabel, "verifying", "verifying runs keep a distinct resume label even though checkpoint.phase stays executing");
132
138
  assert.equal(ids.includes(abandoned.run_id), false);
133
139
  assert.equal(ids.includes(done.run_id), false, "plain done excluded");
134
140
 
@@ -0,0 +1,76 @@
1
+ /** Tests for the stale-extension probe shared by /plans and the completion
2
+ * auditor. */
3
+
4
+ import * as assert from "node:assert/strict";
5
+ import * as fs from "node:fs";
6
+ import * as os from "node:os";
7
+ import * as path from "node:path";
8
+ import { after, before, describe, it } from "node:test";
9
+ import { newestSourceMtime, staleReloadHint, stalenessLine } from "../src/staleness.ts";
10
+
11
+ let root: string;
12
+
13
+ before(() => {
14
+ root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-stale-"));
15
+ fs.mkdirSync(path.join(root, "src"));
16
+ fs.mkdirSync(path.join(root, "tools"));
17
+ fs.writeFileSync(path.join(root, "index.ts"), "export const x = 1;\n");
18
+ fs.writeFileSync(path.join(root, "src", "a.ts"), "export const a = 1;\n");
19
+ });
20
+
21
+ after(() => {
22
+ fs.rmSync(root, { recursive: true, force: true });
23
+ });
24
+
25
+ /** Set every scanned file's mtime to `when`. */
26
+ function touchAll(when: number): void {
27
+ for (const rel of ["index.ts", "src/a.ts", "tools"]) {
28
+ const full = path.join(root, rel);
29
+ if (fs.statSync(full).isDirectory()) {
30
+ fs.writeFileSync(path.join(full, "b.ts"), "export const b = 1;\n");
31
+ fs.utimesSync(path.join(full, "b.ts"), when / 1000, when / 1000);
32
+ } else {
33
+ fs.utimesSync(full, when / 1000, when / 1000);
34
+ }
35
+ }
36
+ }
37
+
38
+ describe("staleness probe", () => {
39
+ it("finds the newest .ts mtime across the root, src and tools", () => {
40
+ const now = Date.now();
41
+ touchAll(now - 60_000);
42
+ const newest = newestSourceMtime(root);
43
+ assert.ok(newest >= now - 61_000 && newest <= now + 1_000, `unexpected mtime ${newest}`);
44
+ });
45
+
46
+ it("reports the copy as current when disk is older than the load time", () => {
47
+ touchAll(Date.now() - 60_000);
48
+ const line = stalenessLine(root, new Date());
49
+ assert.match(line, /up to date/);
50
+ assert.equal(staleReloadHint(root, new Date()), null, "no /reload advice when current");
51
+ });
52
+
53
+ it("reports the copy as stale and advises /reload when disk is newer", () => {
54
+ const future = new Date(Date.now() + 10 * 60_000);
55
+ touchAll(future.getTime());
56
+ const line = stalenessLine(root, new Date(Date.now() - 60_000));
57
+ assert.match(line, /newer than the loaded copy/);
58
+ assert.match(line, /run \/reload/);
59
+ const hint = staleReloadHint(root, new Date(Date.now() - 60_000));
60
+ assert.ok(hint && hint.includes("/reload"), "the hint carries the /reload advice");
61
+ });
62
+
63
+ it("tolerates a same-second write instead of calling it stale", () => {
64
+ // Without the 2s tolerance, a file written moments after load reads as
65
+ // staleness and users chase a /reload that changes nothing.
66
+ touchAll(Date.now());
67
+ const line = stalenessLine(root, new Date());
68
+ assert.match(line, /up to date/);
69
+ });
70
+
71
+ it("degrades to a loaded-timestamp line when the tree cannot be walked", () => {
72
+ const line = stalenessLine(path.join(root, "does-not-exist"), new Date("2026-01-01T00:00:00.000Z"));
73
+ assert.match(line, /Extension loaded: 2026-01-01T00:00:00.000Z/);
74
+ assert.equal(staleReloadHint(path.join(root, "does-not-exist"), new Date()), null);
75
+ });
76
+ });
@@ -263,6 +263,10 @@ describe("runs", () => {
263
263
  const updated = setRunStatus(workdir, run.run_id, "accepted");
264
264
  assert.equal(updated.status, "accepted");
265
265
  assert.equal(getRun(workdir, run.run_id)?.status, "accepted");
266
+ // v0.8: verifying is a valid non-terminal lifecycle status (execution-review loop).
267
+ const verifying = setRunStatus(workdir, run.run_id, "verifying");
268
+ assert.equal(verifying.status, "verifying");
269
+ assert.equal(getRun(workdir, run.run_id)?.status, "verifying");
266
270
  assert.throws(() => setRunStatus(workdir, run.run_id, "bogus"), StateError);
267
271
  });
268
272