pi-plans 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +5 -12
  3. package/README.md +5 -5
  4. package/agents/execution-reviewer.md +92 -0
  5. package/index.ts +13 -23
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +14 -7
  8. package/references/plan-artifact-template.md +11 -1
  9. package/references/state-and-config.md +3 -3
  10. package/scripts/validate.ts +20 -3
  11. package/src/auditor.ts +306 -63
  12. package/src/code-graph/commands.ts +6 -1
  13. package/src/dashboard.ts +91 -13
  14. package/src/exec.ts +835 -142
  15. package/src/plan.ts +1 -1
  16. package/src/refine-ui-state.ts +1 -1
  17. package/src/refine-ui.ts +40 -8
  18. package/src/resume-command.ts +19 -3
  19. package/src/resume.ts +5 -1
  20. package/src/staleness.ts +53 -0
  21. package/src/state.ts +1 -0
  22. package/src/task-tool.ts +1 -1
  23. package/src/tasks.ts +62 -5
  24. package/src/ui-language.ts +4 -0
  25. package/src/workflow-state.ts +93 -6
  26. package/tests/analyze-refs.test.ts +1 -1
  27. package/tests/auditor.test.ts +299 -16
  28. package/tests/dashboard.test.ts +202 -2
  29. package/tests/exec-review-loop.test.ts +724 -0
  30. package/tests/exec.test.ts +198 -44
  31. package/tests/extension-load.test.ts +1 -1
  32. package/tests/refine-ui.test.ts +25 -2
  33. package/tests/resume-lifecycle.test.ts +5 -1
  34. package/tests/resume.test.ts +6 -0
  35. package/tests/staleness.test.ts +76 -0
  36. package/tests/state.test.ts +4 -0
  37. package/tests/tasks.test.ts +142 -0
  38. package/tests/workflow-state.test.ts +105 -0
  39. package/tools/analyze-refs.ts +17 -6
  40. package/tools/execute-plan.ts +12 -5
  41. package/tools/plans.ts +1 -1
  42. package/tools/refine.ts +22 -3
package/src/exec.ts CHANGED
@@ -9,15 +9,16 @@
9
9
  * live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
10
10
  * tree), a stall watchdog pauses the run when consecutive rounds produce no
11
11
  * task-state change, and when every task reaches a terminal state an
12
- * independent completion auditor verifies the plan's verification checks —
13
- * failed checks roll their covered tasks back to pending (audit-flow-only
14
- * channel), and three failed rounds pause for the user (bounded stopped
15
- * termination under auto-approve/headless).
12
+ * independent execution reviewer verifies the plan's verification checks in
13
+ * a detached, overlay-visible loop — failed checks roll their covered tasks
14
+ * back to pending (audit-flow-only channel), and five committed rounds pause
15
+ * the run for the user in every mode (fail-closed, never a silent stop).
16
16
  */
17
17
 
18
18
  import * as fs from "node:fs";
19
19
  import * as path from "node:path";
20
- import { randomUUID } from "node:crypto";
20
+ import { createHash, randomUUID } from "node:crypto";
21
+ import { execSync } from "node:child_process";
21
22
  import type {
22
23
  CompactionResult,
23
24
  ExtensionAPI,
@@ -44,7 +45,7 @@ import {
44
45
  type VccCompactionBuildResult,
45
46
  type VccCompactionStats,
46
47
  } from "./compaction.ts";
47
- import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
48
+ import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, runDirPath, setRunStatus, StateError, utcNow } from "./state.ts";
48
49
  import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
49
50
  import { bindRun, resolveActiveRun } from "./run-context.ts";
50
51
  import type { SubagentProgressEvent } from "./subagent.ts";
@@ -56,6 +57,7 @@ import {
56
57
  applyExecutionProgress,
57
58
  applyExecutionStopped,
58
59
  createCheckpoint,
60
+ applyExecutionPlanAmended,
59
61
  loadCheckpoint,
60
62
  mutateCheckpoint,
61
63
  planIdentityOf,
@@ -71,6 +73,7 @@ import { resolveGraphMode } from "./code-graph/mode.ts";
71
73
  import {
72
74
  parseChecklist,
73
75
  parsePlanTasks,
76
+ CHECKLIST_HEADERS,
74
77
  flattenTasks,
75
78
  type CheckItem,
76
79
  type PlanTasks,
@@ -81,7 +84,10 @@ import {
81
84
  auditableChecks,
82
85
  buildTaskView,
83
86
  currentTask,
87
+ findingsRollbackSet,
84
88
  flattenTaskViews,
89
+ invalidateChecksForRolledBackTasks,
90
+ maxWave,
85
91
  taskIsTerminal,
86
92
  taskProgress,
87
93
  taskProgressMap,
@@ -97,8 +103,15 @@ import {
97
103
  renderDashboardLines,
98
104
  renderDashboardTreeLines,
99
105
  } from "./dashboard.ts";
100
- import { AUDIT_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit } from "./auditor.ts";
106
+ import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
107
+ import type { ReviewFindingRecord } from "./workflow-state.ts";
108
+ import { staleReloadHint as probeStaleReload } from "./staleness.ts";
101
109
  import { messaging } from "./messaging.ts";
110
+ import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
111
+ import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
112
+ import { resolveReviewerSpawn } from "./thinking-levels.ts";
113
+ import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
114
+ import { matchesTerminalKey } from "./terminal-keys.ts";
102
115
 
103
116
  export interface ExecState {
104
117
  planPath: string;
@@ -117,8 +130,14 @@ export interface ExecState {
117
130
  /** Stall watchdog (v0.6.1): consecutive settled rounds without a task
118
131
  * status change; auto-pause at the cap. */
119
132
  stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
120
- /** Completion-audit bookkeeping. */
121
- audit: { rounds: number; failed: string[]; running: boolean };
133
+ /** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
134
+ * only when a round outcome commits (never on discard/cancel) — and is the
135
+ * only piece persisted. */
136
+ audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean; findings: ReviewFinding[] };
137
+ /** Execution-review loop (v0.8), memory-only: the attempt index names the
138
+ * per-round report files; consecutiveDiscards bounds the fingerprint
139
+ * re-run loop; inFlight owns the round's abort lifecycle. */
140
+ review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
122
141
  /** Per-settle audit latch (v0.7.1): a settled round fires the completion
123
142
  * audit at most once, so the turn_end / agent_before_settle / resume entry
124
143
  * points cannot double-consume a round when several land in one settle.
@@ -126,6 +145,18 @@ export interface ExecState {
126
145
  auditLatch?: { auditedThisSettle: boolean; activity: number };
127
146
  }
128
147
 
148
+ /** One in-flight review round: owns its abort lifecycle, its fingerprint of
149
+ * the audited subject, and the per-round one-shot wake token (v0.8). */
150
+ interface InFlightReview {
151
+ controller: AbortController;
152
+ /** Budget round this attempt belongs to (audit.rounds + 1 at spawn). */
153
+ budgetRound: number;
154
+ /** Monotonic attempt ordinal; names the round report file. */
155
+ attempt: number;
156
+ fingerprint: string;
157
+ wakeSent: boolean;
158
+ }
159
+
129
160
  /**
130
161
  * D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
131
162
  * and status bar right after a `set-language` change.
@@ -204,6 +235,10 @@ export interface CheckpointExecutionLoad {
204
235
  /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
205
236
  * detected — resume requires a fresh handoff approval. */
206
237
  legacyDelegate?: boolean;
238
+ /** v0.9.1 (F-005): unresolved findings from the newest committed round,
239
+ * so the /resume-plans brief can surface outstanding highs before the
240
+ * per-turn injection ever runs. */
241
+ findings?: ReviewFinding[];
207
242
  error?: string;
208
243
  }
209
244
 
@@ -253,9 +288,12 @@ export function loadExecutionFromCheckpoint(
253
288
  const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
254
289
  const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
255
290
  const tasks = buildTaskView(planTasks, progress);
256
- // An audit-cap pause grants a fresh audit budget on restore (mirrors
257
- // resumeGoalWaitIfPaused) so the first turn can actually re-audit.
258
- const wasAuditCapPause = (cp.execution.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
291
+ // v0.8: a review-cap pause SURVIVES the restore — the budget must stay
292
+ // bounded across restarts; only /plans-execute grants a fresh one. The
293
+ // legacy v0.7 prefix stays dual-matched for one release.
294
+ const wasReviewCapPause = isReviewCapPause(cp.execution.pausedReason);
295
+ // Any live round from the replaced session graph dies here (CF2-002).
296
+ abortInFlightReview();
259
297
  if (!reverifyAll) {
260
298
  for (const id of cp.execution.doneVcIds) {
261
299
  const item = items.find((candidate) => candidate.id === id);
@@ -271,22 +309,25 @@ export function loadExecutionFromCheckpoint(
271
309
  startedAt: utcNow(),
272
310
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
273
311
  uiLanguage: resolveUiLanguage(ctx.cwd),
274
- // D-020: a paused legacy (or stopped) execution rebuilds unpaused —
275
- // the resume itself is the user's intent; the reason is surfaced in
276
- // the resume brief instead. An audit-cap pause additionally grants a
277
- // fresh audit budget here (mirrors resumeGoalWaitIfPaused), so the
278
- // first turn after a cross-session resume can actually re-audit.
279
- stall: {
280
- rounds: 0,
312
+ // D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
313
+ // resume itself is the user's intent; the reason is surfaced in the
314
+ // resume brief instead. EXCEPT a review-cap pause (v0.8): it must
315
+ // survive restores paused and at its committed round count, or the
316
+ // 5-round budget would never bound anything across restarts.
317
+ stall: {
318
+ rounds: cp.execution.stallRounds ?? 0,
281
319
  lastSnapshot: null,
282
- paused: false,
283
- pausedReason: undefined,
320
+ paused: wasReviewCapPause,
321
+ pausedReason: wasReviewCapPause ? (cp.execution.pausedReason ?? undefined) : undefined,
284
322
  },
285
323
  audit: {
286
- rounds: wasAuditCapPause ? 0 : (cp.execution.audit?.rounds ?? 0),
324
+ rounds: cp.execution.audit?.rounds ?? 0,
287
325
  failed: [],
326
+ undeterminable: cp.execution.audit?.undeterminable ?? [],
327
+ findings: toReviewFindings(cp.execution.audit?.findings),
288
328
  running: false,
289
329
  },
330
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
290
331
  auditLatch: { auditedThisSettle: false, activity: 0 },
291
332
  };
292
333
  execution.stall.lastSnapshot = stallSnapshot();
@@ -295,11 +336,6 @@ export function loadExecutionFromCheckpoint(
295
336
  resetContinuationRuntime(ctx);
296
337
  pendingExecutionFlush = false;
297
338
  resetExecutionCompactionState(ctx);
298
- if (wasAuditCapPause) {
299
- withExecutionCheckpoint(ctx, (cp2) =>
300
- applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
301
- );
302
- }
303
339
  // D-020: an orphaned v0.6.0 delegated executor never survives a restart.
304
340
  // Its checkpoint delegate marker REFUSES the direct load — the run must
305
341
  // re-enter through the execution handoff so the C-006 approval gate
@@ -337,6 +373,7 @@ export function loadExecutionFromCheckpoint(
337
373
  pausedReason: cp.execution.pausedReason,
338
374
  legacyPlan: planTasks.legacy,
339
375
  legacyDelegate,
376
+ findings: toReviewFindings(cp.execution.audit?.findings),
340
377
  };
341
378
  }
342
379
 
@@ -475,6 +512,9 @@ function updatePanelWidget(ctx: ExtensionContext): void {
475
512
  pausedReason: current.stall.pausedReason,
476
513
  auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
477
514
  auditFailed: current.audit.failed,
515
+ auditUndeterminable: current.audit.undeterminable,
516
+ findings: current.audit.findings,
517
+ reviewRunning: current.audit.running === true || current.review.inFlight !== null,
478
518
  startedAt: current.startedAt,
479
519
  usage: current.usage,
480
520
  });
@@ -497,6 +537,9 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
497
537
  pausedReason: execution.stall.pausedReason,
498
538
  auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
499
539
  auditFailed: execution.audit.failed,
540
+ auditUndeterminable: execution.audit.undeterminable,
541
+ findings: execution.audit.findings,
542
+ reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
500
543
  });
501
544
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
502
545
  return;
@@ -524,6 +567,10 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
524
567
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
525
568
  return;
526
569
  }
570
+ if (status === "verifying") {
571
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `🔎 plans: ${active.run_id} (verifying)`));
572
+ return;
573
+ }
527
574
  if (status === "planning") {
528
575
  const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
529
576
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
@@ -564,7 +611,7 @@ function persist(ctx: ExtensionContext): void {
564
611
  startedAt: execution.startedAt,
565
612
  usage: execution.usage,
566
613
  stall: execution.stall,
567
- audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
614
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
568
615
  });
569
616
  }
570
617
 
@@ -594,6 +641,8 @@ export async function startExecution(
594
641
  ctx: ExtensionContext,
595
642
  input: StartExecutionInput,
596
643
  ): Promise<void> {
644
+ // A fresh handoff replaces any live run — abort its in-flight review round first.
645
+ abortInFlightReview();
597
646
  const tasks = buildTaskView(input.planTasks);
598
647
  execution = {
599
648
  planPath: input.planPath,
@@ -605,7 +654,8 @@ export async function startExecution(
605
654
  usage: { inToks: 0, outToks: 0 },
606
655
  uiLanguage: resolveUiLanguage(ctx.cwd),
607
656
  stall: { rounds: 0, lastSnapshot: null, paused: false },
608
- audit: { rounds: 0, failed: [], running: false },
657
+ audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
658
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
609
659
  auditLatch: { auditedThisSettle: false, activity: 0 },
610
660
  };
611
661
  // Seed the watchdog baseline only after `execution` points at the new state
@@ -670,9 +720,14 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
670
720
  applyExecutionProgress(cp, {
671
721
  tasks: taskProgressMap(execution!.tasks),
672
722
  doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
723
+ stallRounds: execution!.stall.rounds,
673
724
  audit: {
674
725
  rounds: execution!.audit.rounds,
675
726
  lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
727
+ undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
728
+ // v0.9: audit writes are replace-semantics — every writer must
729
+ // carry findings or a task update would silently wipe them.
730
+ findings: execution!.audit.findings,
676
731
  },
677
732
  }),
678
733
  );
@@ -699,10 +754,10 @@ export function recordExecutionTurn(
699
754
  }
700
755
 
701
756
  /** Test hook: replace the audit subagent with a deterministic function. */
702
- let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
757
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
703
758
 
704
759
  export function __setAuditRunnerForTests(
705
- runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
760
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
706
761
  ): void {
707
762
  auditRunnerForTests = runner;
708
763
  }
@@ -746,15 +801,17 @@ export function registerExecutionTurnHandlers(
746
801
  maybeContinuationFollowUp(ctx);
747
802
  return;
748
803
  }
749
- // Fully settled and still owed an audit: run it now. The audit's own
750
- // pass/fail message drives the rest (pass completes the run; fail
751
- // triggers a fix turn), so no extra continuation is requested here —
752
- // pendingAudit() is false once paused, which bounds any loop.
804
+ // Fully settled and still owed a review: launch the round under the
805
+ // mode rule — tui/rpc detach (the settle returns NOW and the overlay
806
+ // carries progress); print/json await inline so runtime teardown cannot
807
+ // kill the child. The round's own outcome routing drives the rest.
753
808
  latchAuditThisSettle();
754
- await runAuditFlow(ctx);
809
+ const chain = launchReviewRound(ctx);
810
+ if (chain) await chain;
755
811
  });
756
812
  ext.on("session_shutdown", async (_event, ctx) => {
757
813
  drainExecutionFlush(ctx);
814
+ abortInFlightReview();
758
815
  execution = null;
759
816
  executionRunId = null;
760
817
  continuationRuntime = null;
@@ -795,20 +852,242 @@ export function registerExecutionTurnHandlers(
795
852
  if (usage) recordExecutionTurn(ctx, usage);
796
853
  if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
797
854
  latchAuditThisSettle();
798
- await runAuditFlow(ctx);
855
+ const chain = launchReviewRound(ctx);
856
+ if (chain) await chain;
799
857
  }
800
858
  await onTurnEnd?.(ctx);
801
859
  });
802
860
  }
803
861
 
804
- /** Audit flow: run the completion auditor, apply pass/rollback, and either
805
- * complete the run, keep iterating (rollback), pause at the round cap
806
- * (interactive), or stop at the cap (auto-approve/headless — D-022).
862
+ /** ==== Execution-review loop (v0.8) ====
807
863
  *
808
- * Fail-closed completion: a run completes only when EVERY pending check was
809
- * affirmatively passed (or resolved skipped-pass); checks the auditor failed
810
- * to report count as failed, never as silently passed. */
811
- async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
864
+ * When every task is terminal and checks are still owed, the run enters the
865
+ * `verifying` status and a DETACHED read-only reviewer round runs in the
866
+ * background (tui/rpc): the settle handler returns immediately and the
867
+ * executor is truly idle while the overlay shows live progress. In print/json
868
+ * modes the settle handler keeps AWAITING the round inline — runtime
869
+ * teardown at settle would otherwise kill a detached child and swallow the
870
+ * pause signal.
871
+ *
872
+ * Budget: `audit.rounds` counts COMMITTED rounds only (pass, fail, or
873
+ * undeterminable); discards and cancellations burn nothing. Undeterminable
874
+ * rounds self-schedule the retry inside the loop (no wake). Two consecutive
875
+ * fingerprint discards commit as an undeterminable round so the loop stays
876
+ * bounded. Exhaustion pauses in EVERY mode (fail-closed) with an in-band
877
+ * `pi-plans-review-paused` message; the ONLY fresh-budget surface is
878
+ * /plans-execute — ordinary input and session restores never refill.
879
+ *
880
+ * Lifecycle: each round owns a session-scoped AbortController (never
881
+ * ctx.signal, which is turn-scoped), aborted from session_shutdown,
882
+ * stopExecution, startExecution, and restoreFromSession. Outcomes are
883
+ * guarded by execution identity (`execution !== owner` → silent discard).
884
+ *
885
+ * Completion stays fail-closed AND never fail-open: a run completes only
886
+ * when every pending check was affirmatively passed. */
887
+ const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
888
+ /** v0.7 protocol value — dual-matched for one release so checkpoints written
889
+ * by older builds keep their cap pause recognized on restore. */
890
+ const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
891
+
892
+ function isReviewCapPause(reason: string | undefined | null): boolean {
893
+ if (!reason) return false;
894
+ return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
895
+ }
896
+
897
+ /** The currently-running (or self-scheduling) review chain; the sanctioned
898
+ * test seam awaits this. */
899
+ let activeReviewChain: Promise<void> | null = null;
900
+
901
+ function abortInFlightReview(): void {
902
+ const inFlight = execution?.review.inFlight;
903
+ if (inFlight) {
904
+ try {
905
+ inFlight.controller.abort();
906
+ } catch {
907
+ /* already aborted */
908
+ }
909
+ if (execution) execution.review.inFlight = null;
910
+ }
911
+ activeReviewChain = null;
912
+ }
913
+
914
+ /** v0.9: unresolved high-severity findings from the newest committed round.
915
+ * Presence in the newest round's report IS the unresolved set (stable ids:
916
+ * a problem is resolved only by no longer being reported). */
917
+ function unresolvedHighFindings(ex: ExecState): ReviewFinding[] {
918
+ return ex.audit.findings.filter((f) => f.severity === "high");
919
+ }
920
+
921
+ /** v0.9: checkpoint records -> runtime findings. Invalid severities degrade to
922
+ * "malformed" (recorded, non-blocking) — the same never-fail posture as the
923
+ * report parser. */
924
+ function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
925
+ if (!records) return [];
926
+ return records.map((r) => {
927
+ const severity = (r.severity === "high" || r.severity === "medium" || r.severity === "low") ? r.severity : "malformed";
928
+ return { id: r.id, severity, taskIds: Array.isArray(r.taskIds) ? r.taskIds : [], proposedTask: r.proposedTask, note: r.note ?? "", evidence: r.evidence ?? "", raw: r.raw ?? "" };
929
+ });
930
+ }
931
+
932
+ /** v0.9: findings widen the owed predicate — an unresolved high finding keeps
933
+ * the review owed even when every check is done (the stranded-high path),
934
+ * mirroring how a failed check keeps it owed today. */
935
+ function reviewOwed(ex: ExecState): boolean {
936
+ return allTasksTerminal(ex.tasks)
937
+ && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
938
+ }
939
+
940
+ function runDirOf(ctx: ExtensionContext): string | null {
941
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
942
+ return runId ? runDirPath(ctx.cwd, runId) : null;
943
+ }
944
+
945
+ function setRunStatusForReview(ctx: ExtensionContext, status: "verifying" | "executing"): void {
946
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
947
+ if (!runId) return;
948
+ try {
949
+ const current = getRun(ctx.cwd, runId)?.status;
950
+ if (current !== status && current !== "done" && current !== "abandoned") {
951
+ setRunStatus(ctx.cwd, runId, status);
952
+ }
953
+ } catch {
954
+ /* best-effort */
955
+ }
956
+ }
957
+
958
+ /** Round fingerprint (Q-fingerprint-scope): plan digest + git HEAD +
959
+ * covered-file mtimes. A change between round start and resolve means the
960
+ * reviewer judged a subject that no longer exists — discard and re-run. */
961
+ function captureReviewFingerprint(ctx: ExtensionContext, ex: ExecState): string {
962
+ const parts: string[] = [];
963
+ try {
964
+ parts.push(createHash("sha256").update(fs.readFileSync(ex.planPath, "utf8")).digest("hex"));
965
+ } catch {
966
+ parts.push("plan-unreadable");
967
+ }
968
+ try {
969
+ parts.push(execSync("git rev-parse HEAD", { cwd: ctx.cwd, stdio: ["ignore", "pipe", "pipe"] }).toString().trim());
970
+ } catch {
971
+ parts.push("no-head");
972
+ }
973
+ const covered = new Set<string>();
974
+ for (const task of flattenTaskViews(ex.tasks)) {
975
+ for (const file of task.files ?? []) covered.add(file);
976
+ }
977
+ const mtimes: string[] = [];
978
+ for (const file of [...covered].sort()) {
979
+ try {
980
+ mtimes.push(`${file}:${fs.statSync(path.resolve(ctx.cwd, file)).mtimeMs}`);
981
+ } catch {
982
+ mtimes.push(`${file}:missing`);
983
+ }
984
+ }
985
+ parts.push(mtimes.join("|"));
986
+ return createHash("sha256").update(parts.join("\u0000")).digest("hex");
987
+ }
988
+
989
+ /** Round timeout: a committed round is minutes, never the 60-min subagent
990
+ * default — a hung child must surface as a spawn-failure round, not park the
991
+ * run in verifying for an hour (CF2-010). */
992
+ const REVIEW_ROUND_TIMEOUT_MS = 20 * 60 * 1000;
993
+
994
+ /** Reviewer-role pinning (Q-role-fallback): a CONFIRMED delegated role pins
995
+ * the spawn's model+thinking and labels the overlay with the role; an
996
+ * unconfirmed or current-session role inherits the session default with the
997
+ * "session default" label — a detached round NEVER opens the interactive
998
+ * first-use panel. */
999
+ function reviewSpawnProfile(): { model?: string; thinkingLevel?: string; label: string } {
1000
+ try {
1001
+ const reviewer = loadGlobalConfig().config.reviewer;
1002
+ if (reviewer.mode !== "current-session" && reviewerReady(reviewer)) {
1003
+ const spawn = resolveReviewerSpawn(reviewer);
1004
+ if (spawn.modelSelector) {
1005
+ return { model: spawn.modelSelector, thinkingLevel: spawn.thinkingLevel ?? undefined, label: spawn.label };
1006
+ }
1007
+ }
1008
+ } catch {
1009
+ /* fall through to the session default */
1010
+ }
1011
+ return { label: "session default" };
1012
+ }
1013
+
1014
+ /** Engine-held lane state for the in-flight round: the reopen path builds a
1015
+ * fresh one-shot controller seeded from THIS object, so the accumulated
1016
+ * transcript survives ESC + reopen (CF2-001 / Q-reopen-seed). */
1017
+ let reviewLane: RefineLaneState | null = null;
1018
+ let reviewOverlay: RefineOverlayController | null = null;
1019
+ let reviewModelLabel: string | undefined;
1020
+
1021
+ function freshReviewLane(attempt: number): RefineLaneState {
1022
+ return {
1023
+ id: `review-round-${attempt}`,
1024
+ label: `Execution review round ${attempt}`,
1025
+ status: "queued",
1026
+ phase: "queued",
1027
+ detail: "",
1028
+ transcript: [],
1029
+ currentTurnIndex: 0,
1030
+ scrollOffset: 0,
1031
+ followTranscript: true,
1032
+ viewportHeight: 1,
1033
+ };
1034
+ }
1035
+
1036
+ /** Fresh controller per round (the controller is one-shot: closed latch,
1037
+ * overlayPromise bail, terminal-lane early return — reuse drops progress).
1038
+ * A UI failure must NEVER kill the round itself — best-effort only. The
1039
+ * overlay forwards unhandled keys so Ctrl+Shift+T keeps working while the
1040
+ * review overlay holds focus (pi-tui has no key bubbling). */
1041
+ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
1042
+ if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
1043
+ try {
1044
+ const controller = new RefineOverlayController(
1045
+ "auditor",
1046
+ [{ id: lane.id, label: lane.label }],
1047
+ () => {},
1048
+ lang ?? "en",
1049
+ (data) => {
1050
+ if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
1051
+ },
1052
+ );
1053
+ controller.seedLane(lane);
1054
+ controller.open(refineOverlayContext(ctx), modelLabel);
1055
+ return controller;
1056
+ } catch {
1057
+ return null;
1058
+ }
1059
+ }
1060
+
1061
+ /** The reopen surface (Task-3.4): rebuilds the overlay from engine-held lane
1062
+ * state; inert when no round is in flight. */
1063
+ export function reopenReviewOverlay(ctx: ExtensionContext): void {
1064
+ if (!execution?.review.inFlight || !reviewLane) return;
1065
+ // Never stack a second overlay on a live one (Ctrl+Shift+R while open).
1066
+ if (reviewOverlay && !reviewOverlay.isClosed()) return;
1067
+ const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
1068
+ if (controller) reviewOverlay = controller;
1069
+ }
1070
+
1071
+ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1072
+ const highs = unresolvedHighFindings(ex);
1073
+ const detail =
1074
+ ex.audit.failed.length > 0
1075
+ ? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
1076
+ : highs.length > 0
1077
+ ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1078
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1079
+ const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1080
+ pauseForStall(ctx, reason);
1081
+ // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
1082
+ // pauseForStall's ui.notify is optional and absent headless, so the pause
1083
+ // must also land in the session stream every mode can read.
1084
+ messaging().sendMessage(
1085
+ { customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
1086
+ { triggerTurn: false },
1087
+ );
1088
+ }
1089
+
1090
+ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
812
1091
  if (!execution) return;
813
1092
  const ex = execution;
814
1093
  // Skipped-pass checks resolve without a subagent round.
@@ -819,99 +1098,451 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
819
1098
  // Only auditable checks (with task coverage) gate completion; checks that
820
1099
  // cover no task can never be verified and never block or complete.
821
1100
  const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
822
- if (pendingChecks.length === 0) {
1101
+ // v0.9: unresolved high findings block the fast completion path — they
1102
+ // keep the review owed instead (liveness: a high can never be completed
1103
+ // around, only fixed or paused at the cap).
1104
+ if (pendingChecks.length === 0 && unresolvedHighFindings(ex).length === 0) {
823
1105
  await completeExecution(ctx);
824
1106
  return;
825
1107
  }
826
- if (ex.audit.rounds >= AUDIT_MAX_ROUNDS) {
827
- // D-022: interactive sessions pause for the user (state kept, tasks
828
- // intact, resumable); auto-approve/headless terminates bounded.
829
- if (isInteractiveSession(ctx)) {
830
- pauseForStall(
831
- ctx,
832
- `${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
833
- );
834
- return;
835
- }
836
- await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
1108
+ if (ex.stall.paused || ex.review.inFlight) return;
1109
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1110
+ pauseReviewCap(ctx, ex);
837
1111
  return;
838
1112
  }
839
- ex.audit.running = true;
840
- ex.audit.rounds += 1;
1113
+ // Phase transition: the executor is done with its tasks; the review loop
1114
+ // owns the run until it converges (or pauses at the cap).
1115
+ setRunStatusForReview(ctx, "verifying");
1116
+ const round: InFlightReview = {
1117
+ controller: new AbortController(),
1118
+ budgetRound: ex.audit.rounds + 1,
1119
+ attempt: ex.review.attempts + 1,
1120
+ fingerprint: captureReviewFingerprint(ctx, ex),
1121
+ wakeSent: false,
1122
+ };
1123
+ ex.review.attempts = round.attempt;
1124
+ ex.review.inFlight = round;
1125
+ ex.audit.running = true; // dashboard mirror (the v0.8 model lands with the overlay task)
1126
+ const spawn = reviewSpawnProfile();
1127
+ reviewModelLabel = spawn.label;
1128
+ const lane = freshReviewLane(round.attempt);
1129
+ reviewLane = lane;
1130
+ reviewOverlay = openReviewOverlay(ctx, lane, ex.uiLanguage, spawn.label);
841
1131
  updateStatusWidget(ctx);
842
- let outcome = null as Awaited<ReturnType<typeof runCompletionAudit>>;
1132
+ let result: AuditRoundResult;
843
1133
  try {
844
- outcome = auditRunnerForTests
845
- ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: ex.audit.rounds })
1134
+ result = auditRunnerForTests
1135
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt, priorFindings: ex.audit.findings })
846
1136
  : await runCompletionAudit(ctx, {
847
1137
  planPath: ex.planPath,
848
1138
  checklist: ex.items,
849
1139
  tasks: ex.tasks,
850
- round: ex.audit.rounds,
851
- signal: ctx.signal,
1140
+ round: round.attempt,
1141
+ priorFindings: ex.audit.findings,
1142
+ model: spawn.model,
1143
+ thinkingLevel: spawn.thinkingLevel,
1144
+ timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
1145
+ signal: round.controller.signal,
1146
+ onProgress: (event) => {
1147
+ // The engine owns the lane state; the live controller only repaints.
1148
+ applyRefineProgress(lane, event);
1149
+ reviewOverlay?.rerender();
1150
+ },
1151
+ });
1152
+ } catch (error) {
1153
+ if (ex.review.inFlight === round) {
1154
+ ex.review.inFlight = null;
1155
+ ex.audit.running = false;
1156
+ }
1157
+ void reviewOverlay?.close();
1158
+ reviewOverlay = null;
1159
+ messaging().sendMessage(
1160
+ { customType: "pi-plans-review-error", content: `pi-plans: execution review round threw: ${String(error)}`, display: true },
1161
+ { triggerTurn: false },
1162
+ );
1163
+ updateStatusWidget(ctx);
1164
+ return;
1165
+ }
1166
+ // Overlay terminal state + close (the controller is one-shot; the engine-held
1167
+ // lane keeps the transcript for a later reopen within this round).
1168
+ const cancelledResult = result !== null && typeof result === "object" && "cancelled" in result;
1169
+ try {
1170
+ applyRefineResult(lane, {
1171
+ ok: !cancelledResult,
1172
+ output: result && "report" in result ? result.report : "",
1173
+ stderr: "",
1174
+ turns: 0,
1175
+ ...(cancelledResult ? { cancelled: true as const } : {}),
1176
+ });
1177
+ } catch {
1178
+ /* cosmetic only */
1179
+ }
1180
+ void reviewOverlay?.close();
1181
+ reviewOverlay = null;
1182
+ await handleReviewOutcome(ctx, ex, round, result, pendingChecks.map((item) => item.id));
1183
+ }
1184
+
1185
+ async function handleReviewOutcome(
1186
+ ctx: ExtensionContext,
1187
+ owner: ExecState,
1188
+ round: InFlightReview,
1189
+ result: AuditRoundResult,
1190
+ pendingIds: string[],
1191
+ ): Promise<void> {
1192
+ // Identity guard: a restore, stop, or fresh handoff replaced the run —
1193
+ // drop this outcome silently (CF2-002).
1194
+ if (execution !== owner) return;
1195
+ if (owner.review.inFlight !== round) return;
1196
+ owner.review.inFlight = null;
1197
+ owner.audit.running = false;
1198
+ if (result !== null && typeof result === "object" && "cancelled" in result) {
1199
+ // Aborted by shutdown/stop/restore/tree-switch: no round, no budget, no wake.
1200
+ updateStatusWidget(ctx);
1201
+ return;
1202
+ }
1203
+ const outcome = result;
1204
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1205
+ const fingerprintNow = captureReviewFingerprint(ctx, owner);
1206
+ const coveredTaskIds = flattenTaskViews(owner.tasks).map((task) => task.id);
1207
+ if (fingerprintNow !== round.fingerprint) {
1208
+ // The audited subject moved under the reviewer (Q-C): discard, re-run
1209
+ // without budget burn; two consecutive discards commit as undeterminable
1210
+ // so a mutating user cannot loop the loop for free (Q-discard-bound).
1211
+ owner.review.consecutiveDiscards += 1;
1212
+ const runDir = runDirOf(ctx);
1213
+ if (runDir) {
1214
+ writeReviewRoundReport(runDir, {
1215
+ budgetRound: round.budgetRound,
1216
+ attempt: round.attempt,
1217
+ outcome: "discarded",
1218
+ passed: [],
1219
+ failed: [],
1220
+ undeterminable: pendingIds,
1221
+ // v0.9.1 (F-014): the discard report must not understate a
1222
+ // round whose embedded report carries highs.
1223
+ findings: owner.audit.findings,
1224
+ discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
1225
+ fingerprintCaptured: round.fingerprint,
1226
+ fingerprintFound: fingerprintNow,
1227
+ coveredTaskIds,
1228
+ report: reportText,
1229
+ });
1230
+ }
1231
+ if (owner.review.consecutiveDiscards >= 2) {
1232
+ owner.review.consecutiveDiscards = 0;
1233
+ await commitReviewOutcome(
1234
+ ctx,
1235
+ owner,
1236
+ round,
1237
+ { passed: [], failed: [], undeterminable: pendingIds, report: `(two consecutive fingerprint discards — committed as an undeterminable round)\n\n${reportText}` },
1238
+ pendingIds,
1239
+ );
1240
+ return;
1241
+ }
1242
+ persist(ctx);
1243
+ updateStatusWidget(ctx);
1244
+ await maybeContinueReview(ctx, owner);
1245
+ return;
1246
+ }
1247
+ owner.review.consecutiveDiscards = 0;
1248
+ await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
1249
+ }
1250
+
1251
+ /** v0.9 (findings-driven fix loop): append plan tasks for high findings no
1252
+ * existing task owns. The reviewer stays read-only — it proposes the title
1253
+ * (proposed-task); this machinery applies it with provenance, so every high
1254
+ * finding always has an owner the executor can close, and the stable F-###
1255
+ * id rides in the task title for traceability. Best-effort: an unwritable
1256
+ * plan must not crash the loop (the finding then stays stranded and the cap
1257
+ * pause surfaces it). Returns the appended task ids. */
1258
+ function appendFindingTasks(ex: ExecState, highs: ReviewFinding[]): string[] {
1259
+ const appended: string[] = [];
1260
+ let next = flattenTaskViews(ex.tasks).reduce((max, t) => {
1261
+ const m = /^Task-(\d+)$/.exec(t.id);
1262
+ return m ? Math.max(max, Number(m[1])) : max;
1263
+ }, 0);
1264
+ const wave = maxWave(ex.tasks) + 1;
1265
+ try {
1266
+ const planText = fs.readFileSync(ex.planPath, "utf8");
1267
+ const lines = planText.split("\n");
1268
+ // Insert inside the Tasks section: before the Execution Waves
1269
+ // subsection when present, else before the FIRST checklist header the
1270
+ // plan uses (v0.9.1 F-013: legacy plans say `## Verifier Checklist`,
1271
+ // and an EOF fallback would land outside every parsed section), else
1272
+ // at EOF; walk back over blank separators so the bullet lands
1273
+ // adjacent to its siblings.
1274
+ let insertAt = lines.length;
1275
+ const wavesIdx = lines.findIndex((l) => /^###\s+Execution Waves/.test(l));
1276
+ const checklistIdx = lines.findIndex((l) => CHECKLIST_HEADERS.some((h) => new RegExp(`^##\\s+${h}`).test(l)));
1277
+ if (wavesIdx !== -1) insertAt = wavesIdx;
1278
+ else if (checklistIdx !== -1) insertAt = checklistIdx;
1279
+ while (insertAt > 0 && lines[insertAt - 1].trim() === "") insertAt--;
1280
+ // v0.9.1 (F-012): reviewer text becomes task-title metadata at parse
1281
+ // time — strip the microsyntax metacharacters (em/en dashes, `--`
1282
+ // separators, the `;` field delimiter) so an embedded token can never
1283
+ // split title from tail or forge fields.
1284
+ const sanitize = (text: string): string => text.replace(/[—–]/g, "-").replace(/-{2,}/g, "-").replace(/;/g, ",");
1285
+ const entries: Array<{ id: string; title: string }> = [];
1286
+ const newLines: string[] = [];
1287
+ for (const h of highs) {
1288
+ next += 1;
1289
+ const id = `Task-${next}`;
1290
+ const title = `fix ${h.id}: ${sanitize(h.proposedTask ?? h.note ?? "address the finding")} (appended by execution review round ${ex.audit.rounds})`;
1291
+ // v0.9.1 (F-006): carry the wave in the bullet tail so a re-parse
1292
+ // restores the same wave the live tree assigned — without it the
1293
+ // appended remediation task fell back to wave 1 on restore and
1294
+ // hijacked the ▸ anchor.
1295
+ newLines.push(`- \`${id}\`: ${title} — wave: ${wave}`);
1296
+ entries.push({ id, title });
1297
+ }
1298
+ lines.splice(insertAt, 0, ...newLines);
1299
+ fs.writeFileSync(ex.planPath, lines.join("\n"), "utf8");
1300
+ // v0.9.1 (F-008): only a successful plan write mints the live tasks —
1301
+ // pushing before the write left checkpoint entries the plan file does
1302
+ // not contain whenever the write failed.
1303
+ for (const entry of entries) {
1304
+ ex.tasks.push({ id: entry.id, title: entry.title, wave, deps: [], files: [], status: "pending", children: [] });
1305
+ appended.push(entry.id);
1306
+ }
1307
+ } catch {
1308
+ /* best-effort: stranded highs surface via the cap pause */
1309
+ }
1310
+ return appended;
1311
+ }
1312
+
1313
+ async function commitReviewOutcome(
1314
+ ctx: ExtensionContext,
1315
+ ex: ExecState,
1316
+ round: InFlightReview,
1317
+ outcome: AuditOutcome | null,
1318
+ pendingIds: string[],
1319
+ ): Promise<void> {
1320
+ // The budget is charged only when an outcome commits — never on discard
1321
+ // or cancellation (CF2-003).
1322
+ ex.audit.rounds = round.budgetRound;
1323
+ const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
1324
+ const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
1325
+ // Anything the round neither passed nor failed is undeterminable: the
1326
+ // report omitted the check, spelled the verdict unreadably, or the
1327
+ // subagent never ran.
1328
+ const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
1329
+ // v0.9: the newest round's reported findings ARE the unresolved set
1330
+ // (stable ids — a problem is resolved only by no longer being reported).
1331
+ // v0.9.1 (F-001): only a REAL parsed report is authoritative. A round that
1332
+ // produced no report at all — spawn failure (outcome === null) or the
1333
+ // two-consecutive-discard synthesis — must PRESERVE the unresolved set:
1334
+ // clearing it let the vacuous completion guard (empty pendingIds) mark a
1335
+ // run done with its high finding silently dropped.
1336
+ const reported = outcome !== null && outcome.findings !== undefined;
1337
+ const findings = reported ? outcome.findings : ex.audit.findings;
1338
+ ex.audit.findings = findings;
1339
+ const highs = findings.filter((f) => f.severity === "high");
1340
+ // Findings are actionable only when this round actually reported them;
1341
+ // see the fix-loop branch below (v0.9.1, F-001).
1342
+ const actionableHighs = reported ? highs : [];
1343
+ const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
1344
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1345
+ let reportPath: string | null = null;
1346
+ {
1347
+ const runDir = runDirOf(ctx);
1348
+ if (runDir) {
1349
+ reportPath = writeReviewRoundReport(runDir, {
1350
+ budgetRound: round.budgetRound,
1351
+ attempt: round.attempt,
1352
+ outcome: outcome === null
1353
+ ? "spawn-failed"
1354
+ : failed.length > 0 || actionableHighs.length > 0
1355
+ ? "failed"
1356
+ : undeterminable.length > 0 ? "undeterminable" : "passed",
1357
+ passed,
1358
+ failed,
1359
+ undeterminable,
1360
+ findings,
1361
+ fingerprintCaptured: round.fingerprint,
1362
+ coveredTaskIds,
1363
+ report: reportText,
1364
+ });
1365
+ }
1366
+ }
1367
+ // Partial progress counts: a check affirmed this round is done even when
1368
+ // a sibling failed, so a later round only re-judges what is still open.
1369
+ for (const id of passed) {
1370
+ const item = ex.items.find((candidate) => candidate.id === id);
1371
+ if (item) item.done = true;
1372
+ }
1373
+
1374
+ // v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
1375
+ // high finding can never complete the run (liveness). Failed checks and
1376
+ // high findings drive ONE union rollback and exactly one executor wake;
1377
+ // high wins over the undeterminable self-schedule (a finding is
1378
+ // actionable independent of verdict evidence). v0.9.1 (F-001): verdicts
1379
+ // are authoritative whenever a report exists, but the FINDINGS-driven
1380
+ // half of the branch needs a findings-bearing report — a no-findings
1381
+ // round (spawn failure, discard synthesis, legacy shape) preserves the
1382
+ // unresolved set and self-schedules instead of rolling back on it.
1383
+ if (failed.length > 0 || actionableHighs.length > 0) {
1384
+ ex.audit.failed = failed;
1385
+ ex.audit.undeterminable = undeterminable;
1386
+ const rolledBack: string[] = [];
1387
+ for (const id of failed) {
1388
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1389
+ }
1390
+ // Invalidation is the VC-fail invariant ONLY: a check re-verifies its
1391
+ // reopened tasks when a FAILED check rolled them back. A pure
1392
+ // finding-driven rollback keeps earlier passes — the findings channel
1393
+ // itself re-examines the repaired work next round (stable ids).
1394
+ if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1395
+ const knownIds = new Set(coveredTaskIds);
1396
+ const mappedHighIds = [...new Set(actionableHighs.flatMap((h) => h.taskIds).filter((id) => knownIds.has(id)))];
1397
+ const highRolledBack = findingsRollbackSet(ex.tasks, mappedHighIds);
1398
+ const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
1399
+ const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
1400
+ const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
1401
+ withExecutionCheckpoint(ctx, (cp) => {
1402
+ // v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
1403
+ // re-stamp the checkpoint's plan identity in the same revision so a
1404
+ // later /resume-plans accepts the amended plan instead of rejecting
1405
+ // it as plan-mismatch (which would cost a full re-approval).
1406
+ const amendedCp = amended.length > 0
1407
+ ? applyExecutionPlanAmended(cp, planIdentityOf(ex.planPath, cp.plan?.version ?? 1), ex.audit.rounds)
1408
+ : cp;
1409
+ return applyExecutionProgress(amendedCp, {
1410
+ tasks: taskProgressMap(ex.tasks),
1411
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1412
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
852
1413
  });
853
- } finally {
854
- ex.audit.running = false;
855
- }
856
- // Fail-closed: checks the outcome did not affirmatively pass are failed.
857
- const reportedPass = new Set(outcome?.passed ?? []);
858
- const failed = pendingChecks
859
- .map((item) => item.id)
860
- .filter((id) => !reportedPass.has(id));
861
- if (failed.length === 0) {
1414
+ });
1415
+ if (allRolledBack.length > 0 || amended.length > 0) {
1416
+ // Rolling back (or appending) is itself forward progress for the
1417
+ // watchdog, but NOT for audit.rounds: that counter stays monotonic
1418
+ // so the repair loop is bounded. The repair belongs to the
1419
+ // executor — back to executing.
1420
+ ex.stall.rounds = 0;
1421
+ ex.stall.lastSnapshot = stallSnapshot();
1422
+ setRunStatusForReview(ctx, "executing");
1423
+ }
1424
+ persist(ctx);
1425
+ updateStatusWidget(ctx);
1426
+ // v0.8 wake, generalized (v0.9): per-round one-shot token — exactly one
1427
+ // triggerTurn per committed fix-needing outcome, even when the round
1428
+ // resolves long after the settle that spawned it.
1429
+ if (!round.wakeSent) {
1430
+ round.wakeSent = true;
1431
+ const openTasks = flattenTaskViews(ex.tasks)
1432
+ .filter((task) => !taskIsTerminal(task))
1433
+ .map((task) => task.id);
1434
+ const stranded = allRolledBack.length === 0 && amended.length === 0;
1435
+ const highLines = actionableHighs.map((h) => `- ${h.id}${h.taskIds.length ? ` (${h.taskIds.join(", ")})` : ""}: ${h.note}`).join("\n");
1436
+ const reportRef = reportPath
1437
+ ? `Full round report: ${reportPath}`
1438
+ : `Full round report (run dir unwritable — inline):\n\n---\n${reportText.slice(0, 4000)}`;
1439
+ // v0.9.1 (F-004): a pure VC-fail round keeps the v0.8 lead — never
1440
+ // announce "0 high-severity findings" over an empty block.
1441
+ const findingsLead = actionableHighs.length > 0
1442
+ ? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
1443
+ : `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
1444
+ const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
1445
+ const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
1446
+ messaging().sendMessage(
1447
+ {
1448
+ customType: "pi-plans-audit-failed",
1449
+ content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}`,
1450
+ display: true,
1451
+ },
1452
+ { triggerTurn: true },
1453
+ );
1454
+ }
1455
+ return; // The agent repairs; the next settle re-enters the loop.
1456
+ }
1457
+
1458
+ // Fail-closed completion: every pending check affirmatively passed AND no
1459
+ // high finding remains. The highs guard is explicit (v0.9.1, F-001): a
1460
+ // no-report round skips the fix branch above, so this is the last line
1461
+ // against vacuously completing an empty-pendingIds round with an
1462
+ // unresolved high. An all-undeterminable round yields failed === [] —
1463
+ // completing here would be fail-open, marking a run done with nothing
1464
+ // verified.
1465
+ if (passed.length === pendingIds.length && highs.length === 0) {
862
1466
  ex.audit.failed = [];
1467
+ ex.audit.undeterminable = [];
863
1468
  withExecutionCheckpoint(ctx, (cp) =>
864
1469
  applyExecutionProgress(cp, {
865
1470
  tasks: taskProgressMap(ex.tasks),
866
1471
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
867
- audit: { rounds: ex.audit.rounds, passed: true },
1472
+ audit: { rounds: ex.audit.rounds, passed: true, findings },
868
1473
  }),
869
1474
  );
870
1475
  await completeExecution(ctx);
871
1476
  return;
872
1477
  }
873
- // Failed checks: roll their covered tasks back (always — an unreported or
874
- // infra-failed audit must reopen work so the loop can continue) and
875
- // persist progress including checks that passed earlier rounds.
876
- for (const id of failed) {
877
- const item = ex.items.find((candidate) => candidate.id === id);
878
- if (item) item.done = false;
879
- }
1478
+
1479
+ // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1480
+ // wake, no message; the dashboard/overlay carries the round counter.
880
1481
  ex.audit.failed = failed;
881
- const rolledBack: string[] = [];
882
- for (const id of failed) {
883
- rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
884
- }
1482
+ ex.audit.undeterminable = undeterminable;
885
1483
  withExecutionCheckpoint(ctx, (cp) =>
886
1484
  applyExecutionProgress(cp, {
887
1485
  tasks: taskProgressMap(ex.tasks),
888
1486
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
889
- audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
1487
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
890
1488
  }),
891
1489
  );
892
- ex.stall.rounds = 0;
893
- ex.stall.lastSnapshot = stallSnapshot();
894
1490
  persist(ctx);
895
1491
  updateStatusWidget(ctx);
896
- const report = outcome?.report ?? "(audit subagent failed to run)";
897
- // v0.7.1: the failure notification now wakes the agent so it can fix the
898
- // rolled-back work without the user having to poke the run (root cause of
899
- // the observed stall). The wake is the ONLY turn this settle emits — the
900
- // rollback below also marks the runtime handled so the continuation path
901
- // cannot add a second EXECUTION_CONTINUE wake for the same settle (Q-2).
902
- const stranded = rolledBack.length === 0;
903
- messaging().sendMessage(
904
- {
905
- customType: "pi-plans-audit-failed",
906
- content: `**pi-plans: completion audit round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the audit reruns automatically once all tasks are terminal again.${ex.audit.rounds >= AUDIT_MAX_ROUNDS ? ` This was round ${AUDIT_MAX_ROUNDS} of ${AUDIT_MAX_ROUNDS}: interactive sessions pause for review; the next terminal-task cycle stops or pauses the run.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the audit automatically.` : ""}\n\n---\n${report.slice(0, 4000)}`,
907
- display: true,
908
- },
909
- { triggerTurn: true },
910
- );
911
- // Q-2 (single wake): this message owns the settle's only turn. Mark the
912
- // runtime handled so maybeContinuationFollowUp stays silent.
913
- const runtime = currentContinuationRuntime(ctx);
914
- if (runtime) runtime.handled = true;
1492
+ await maybeContinueReview(ctx, ex);
1493
+ }
1494
+
1495
+ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promise<void> {
1496
+ if (execution !== ex) return;
1497
+ if (!reviewOwed(ex)) return;
1498
+ if (ex.stall.paused) return;
1499
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1500
+ pauseReviewCap(ctx, ex);
1501
+ return;
1502
+ }
1503
+ await startReviewRound(ctx);
1504
+ }
1505
+
1506
+ /** Launch a review round under the mode rule: detached (fire-and-forget with
1507
+ * the chain tracked for the test seam) in tui/rpc; awaited inline otherwise
1508
+ * (print/json settle must hold the runtime open through the round). Returns
1509
+ * the chain when the caller must await it, null when detached. */
1510
+ function launchReviewRound(ctx: ExtensionContext): Promise<void> | null {
1511
+ const detach = ctx.mode === "tui" || ctx.mode === "rpc";
1512
+ const chain = (async () => {
1513
+ await startReviewRound(ctx);
1514
+ })();
1515
+ activeReviewChain = chain;
1516
+ if (detach) {
1517
+ chain.catch(() => {
1518
+ /* surfaced via the review messages */
1519
+ });
1520
+ return null;
1521
+ }
1522
+ return chain;
1523
+ }
1524
+
1525
+ /** Deterministic seam: await the in-flight (or self-scheduling) review chain. */
1526
+ export function __awaitReviewRoundForTests(): Promise<void> {
1527
+ return activeReviewChain ?? Promise.resolve();
1528
+ }
1529
+
1530
+ /** When this module graph was first imported into the running pi process.
1531
+ * pi loads extensions once, so a fix written to disk mid-session stays
1532
+ * invisible until /reload — which is exactly why an unreadable audit verdict
1533
+ * deserves a /reload hint rather than a bare retry. */
1534
+ const extensionModuleLoadedAt = new Date();
1535
+
1536
+ /** The /reload advice when the extension on disk is newer than the copy this
1537
+ * process loaded, else null. Shared with /plans via src/staleness.ts so both
1538
+ * report the same answer from one probe. */
1539
+ function staleReloadHint(): string | null {
1540
+ try {
1541
+ const root = path.dirname(path.dirname(new URL(import.meta.url).pathname));
1542
+ return probeStaleReload(root, extensionModuleLoadedAt);
1543
+ } catch {
1544
+ return null;
1545
+ }
915
1546
  }
916
1547
 
917
1548
  /** True when the session can surface a pause to a human (D-022): interactive
@@ -931,7 +1562,7 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
931
1562
  const run = getRun(ctx.cwd, active.run_id);
932
1563
  if (!run) return null;
933
1564
  if (phase === "planning" && run.status !== "planning") return null;
934
- if (phase === "execution" && run.status !== "executing") return null;
1565
+ if (phase === "execution" && run.status !== "executing" && run.status !== "verifying") return null;
935
1566
  scaffoldVccSettings(stateRoot);
936
1567
  return { settings: loadVccSettings(stateRoot), runId: run.run_id, artifactDir: run.artifact_dir };
937
1568
  }
@@ -1392,6 +2023,9 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
1392
2023
 
1393
2024
  export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
1394
2025
  if (!execution) return;
2026
+ // A stopped run's in-flight review round dies with it (typed cancelled —
2027
+ // no budget, no wake).
2028
+ abortInFlightReview();
1395
2029
  resetExecutionCompactionState(ctx);
1396
2030
  pendingExecutionFlush = false;
1397
2031
  persist(ctx);
@@ -1437,7 +2071,7 @@ function stallSnapshot(): string {
1437
2071
  }
1438
2072
 
1439
2073
  /**
1440
- * v0.7.1: shared "the completion audit is owed" predicate. Every entry point
2074
+ * v0.7.1: shared "the execution review is owed" predicate. Every entry point
1441
2075
  * that can start the audit (turn_end, agent_before_settle, restoreFromSession,
1442
2076
  * the resume path) goes through this so they can never disagree.
1443
2077
  *
@@ -1446,12 +2080,13 @@ function stallSnapshot(): string {
1446
2080
  */
1447
2081
  function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1448
2082
  if (!ex) return false;
1449
- // A paused run is never self-driven: the stall / audit-cap pause is an
2083
+ // A paused run is never self-driven: the stall / review-cap pause is an
1450
2084
  // explicit "hand control back" signal, and resuming it is the user's call.
1451
2085
  // This also bounds the zero-input continue loop in agent_before_settle.
1452
2086
  if (ex.stall.paused) return false;
1453
- if (ex.audit.running) return false;
1454
- return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
2087
+ if (ex.review.inFlight) return false;
2088
+ return allTasksTerminal(ex.tasks)
2089
+ && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
1455
2090
  }
1456
2091
 
1457
2092
  /** v0.7.1: record that this settle already ran (or declined) its audit, so a
@@ -1564,48 +2199,78 @@ export function filterContinuationMessages<T extends { customType?: string; deta
1564
2199
  return filterGoalWaitMessages(messages);
1565
2200
  }
1566
2201
 
1567
- /** Prefix of the stall reason used for the audit-cap pause (D-022). */
1568
- const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
1569
-
1570
2202
  /** Called for genuine user input or an explicit same-execution resume.
1571
- * Resuming an audit-cap pause grants a fresh audit budget (three more
1572
- * rounds): the user's explicit resume IS the decision to keep auditing —
1573
- * without this reset the cap pause could never be lifted productively. */
2203
+ * v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
2204
+ * refill the five-round budget (CF2-004); only /plans-execute
2205
+ * (resumeActiveExecution) is the explicit confirmation surface. Genuine
2206
+ * stall pauses still clear on input as before. */
1574
2207
  export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
1575
2208
  const ex = getExecution();
1576
2209
  if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
1577
- const wasAuditCap = (ex.stall.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
2210
+ if (isReviewCapPause(ex.stall.pausedReason)) {
2211
+ // Surfaced once per input so the user is not left guessing why the run
2212
+ // stays paused; the pause itself and the budget survive untouched.
2213
+ ctx.ui.notify?.(
2214
+ "pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
2215
+ "warning",
2216
+ );
2217
+ return false;
2218
+ }
1578
2219
  ex.stall.paused = false;
1579
2220
  ex.stall.pausedReason = undefined;
1580
2221
  ex.stall.rounds = 0;
1581
2222
  ex.stall.lastSnapshot = stallSnapshot();
1582
- if (wasAuditCap) {
1583
- ex.audit.rounds = 0;
1584
- ex.audit.failed = [];
1585
- withExecutionCheckpoint(ctx, (cp) =>
1586
- applyExecutionProgress(cp, {
1587
- tasks: taskProgressMap(ex.tasks),
1588
- audit: { rounds: 0, lastResult: undefined },
1589
- pausedReason: null,
1590
- }),
1591
- );
1592
- } else {
1593
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
1594
- }
2223
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
1595
2224
  persist(ctx);
1596
2225
  updateStatusWidget(ctx);
1597
2226
  return true;
1598
2227
  }
1599
2228
 
1600
2229
  export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2230
+ // v0.8: /plans-execute is THE explicit confirmation surface for a
2231
+ // review-cap pause — the only place a fresh five-round budget is granted
2232
+ // (Q-confirm-surface). Ordinary input and session restores never refill.
2233
+ const pausedEx = getExecution();
2234
+ if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
2235
+ pausedEx.stall.paused = false;
2236
+ pausedEx.stall.pausedReason = undefined;
2237
+ pausedEx.stall.rounds = 0;
2238
+ pausedEx.stall.lastSnapshot = stallSnapshot();
2239
+ pausedEx.audit.rounds = 0;
2240
+ pausedEx.audit.failed = [];
2241
+ pausedEx.audit.undeterminable = [];
2242
+ // v0.9: the fresh budget inherits unresolved findings (stable ids keep
2243
+ // counting) — only the round counter resets.
2244
+ withExecutionCheckpoint(ctx, (cp) =>
2245
+ applyExecutionProgress(cp, {
2246
+ tasks: taskProgressMap(pausedEx.tasks),
2247
+ audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
2248
+ pausedReason: null,
2249
+ }),
2250
+ );
2251
+ persist(ctx);
2252
+ updateStatusWidget(ctx);
2253
+ messaging().sendMessage(
2254
+ {
2255
+ customType: "pi-plans-review-budget-granted",
2256
+ content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
2257
+ display: true,
2258
+ },
2259
+ { triggerTurn: false },
2260
+ );
2261
+ const grantChain = launchReviewRound(ctx);
2262
+ if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
2263
+ return true;
2264
+ }
1601
2265
  // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
1602
2266
  // to `return false` here, so `/plans-execute` answered "already executing"
1603
2267
  // and the run stayed stranded until a full re-entry or a session restore.
1604
2268
  // It is not paused, so the pause path below cannot see it — check it first
1605
- // and run the owed audit instead of reporting "nothing to resume".
2269
+ // and run the owed review instead of reporting "nothing to resume".
1606
2270
  if (!execution?.stall.paused && pendingAudit()) {
1607
2271
  latchAuditThisSettle();
1608
- void runAuditFlow(ctx).catch(() => { /* surfaced via the audit message */ });
2272
+ const chain = launchReviewRound(ctx);
2273
+ if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
1609
2274
  return true;
1610
2275
  }
1611
2276
  if (!resumeGoalWaitIfPaused(ctx)) return false;
@@ -1626,6 +2291,12 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1626
2291
  const summary = flat
1627
2292
  .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
1628
2293
  .join("\n");
2294
+ // v0.9: residual (non-high) findings are summarized, never silent — the
2295
+ // loop converged because nothing high-blocking remained.
2296
+ const residualFindings = execution.audit.findings.filter((f) => f.severity !== "high");
2297
+ const residualNote = residualFindings.length > 0
2298
+ ? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
2299
+ : "";
1629
2300
  const planPath = execution.planPath;
1630
2301
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
1631
2302
  execution = null;
@@ -1635,7 +2306,7 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1635
2306
  messaging().sendMessage(
1636
2307
  {
1637
2308
  customType: "pi-plans-complete",
1638
- content: `**Plan complete!** ✅ \`${planPath}\` — completion audit passed.\n\n${summary}`,
2309
+ content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
1639
2310
  display: true,
1640
2311
  },
1641
2312
  { triggerTurn: false },
@@ -1668,7 +2339,11 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
1668
2339
  ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
1669
2340
  : graphBlockForExecutor(mode === "enabled");
1670
2341
  const rollbackNote = execution.audit.failed.length > 0
1671
- ? `\nCompletion audit round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
2342
+ ? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
2343
+ : "";
2344
+ const unresolvedHighs = unresolvedHighFindings(execution);
2345
+ const highFindingsNote = unresolvedHighs.length > 0
2346
+ ? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
1672
2347
  : "";
1673
2348
  return `[PI-PLANS EXECUTION — write access enabled]
1674
2349
  Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
@@ -1676,7 +2351,7 @@ Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${p
1676
2351
  Current wave ${currentWave} open tasks:
1677
2352
  ${waveList}
1678
2353
 
1679
- Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
2354
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}
1680
2355
 
1681
2356
  ${graphLine}
1682
2357
 
@@ -1684,7 +2359,7 @@ Execution rules:
1684
2359
  - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
1685
2360
  - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
1686
2361
  - Close subtasks before their parent; a parent is auditable only when every child is terminal.
1687
- - When every task is terminal, the independent completion auditor verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
2362
+ - When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
1688
2363
  - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1689
2364
  - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
1690
2365
  - Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
@@ -1741,10 +2416,18 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1741
2416
  planTasks = parsePlanTasks(planText);
1742
2417
  const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
1743
2418
  tasks = buildTaskView(planTasks, snapshotProgress);
1744
- items = parseChecklist(planText);
2419
+ // Fresh text wins, but the satisfied state survives the re-parse: a
2420
+ // mid-loop session restore (v0.9.1, found via F-001's self-schedule
2421
+ // path) must not hand the next round a brief that re-judges checks an
2422
+ // earlier round already passed — that burned budget on every /reload.
2423
+ const snapDone = new Set(snapshot.items.filter((c) => c.done).map((c) => c.id));
2424
+ items = parseChecklist(planText).map((item) => (snapDone.has(item.id) ? { ...item, done: true } : item));
1745
2425
  } catch {
1746
2426
  tasks = snapshot.tasks;
1747
2427
  }
2428
+ // The snapshot cannot carry an in-flight round (it is memory-only); abort
2429
+ // any live one from the previous session graph and rebuild fresh (CF2-002).
2430
+ abortInFlightReview();
1748
2431
  execution = {
1749
2432
  planPath: snapshot.planPath,
1750
2433
  items,
@@ -1754,8 +2437,15 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1754
2437
  startedAt: snapshot.startedAt,
1755
2438
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
1756
2439
  uiLanguage: resolveUiLanguage(ctx.cwd),
1757
- stall: { ...snapshot.stall, lastSnapshot: null, rounds: 0 },
1758
- audit: { rounds: snapshot.audit?.rounds ?? 0, failed: snapshot.audit?.failed ?? [], running: false },
2440
+ stall: { ...snapshot.stall, lastSnapshot: null },
2441
+ audit: {
2442
+ rounds: snapshot.audit?.rounds ?? 0,
2443
+ failed: snapshot.audit?.failed ?? [],
2444
+ undeterminable: [],
2445
+ findings: toReviewFindings(snapshot.audit?.findings),
2446
+ running: false,
2447
+ },
2448
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
1759
2449
  auditLatch: { auditedThisSettle: false, activity: 0 },
1760
2450
  };
1761
2451
  execution.stall.lastSnapshot = stallSnapshot();
@@ -1765,9 +2455,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1765
2455
  if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1766
2456
  persist(ctx);
1767
2457
  if (pendingAudit()) {
1768
- // Terminal tasks without a passing audit: rerun the audit flow.
2458
+ // Self-heal (v0.8): a verifying run with pending checks restarts its
2459
+ // round exactly once per resume — the full predicate (paused, in-flight,
2460
+ // budget) lives inside startReviewRound. Restores never grant budget.
1769
2461
  latchAuditThisSettle();
1770
- await runAuditFlow(ctx);
2462
+ const chain = launchReviewRound(ctx);
2463
+ if (chain) await chain;
1771
2464
  }
1772
2465
  updateStatusWidget(ctx);
1773
2466
  }