pi-plans 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/exec.ts CHANGED
@@ -9,15 +9,16 @@
9
9
  * live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
10
10
  * tree), a stall watchdog pauses the run when consecutive rounds produce no
11
11
  * task-state change, and when every task reaches a terminal state an
12
- * independent completion auditor verifies the plan's verification checks —
13
- * failed checks roll their covered tasks back to pending (audit-flow-only
14
- * channel), and three failed rounds pause for the user (bounded stopped
15
- * termination under auto-approve/headless).
12
+ * independent execution reviewer verifies the plan's verification checks in
13
+ * a detached, overlay-visible loop — failed checks roll their covered tasks
14
+ * back to pending (audit-flow-only channel), and five committed rounds pause
15
+ * the run for the user in every mode (fail-closed, never a silent stop).
16
16
  */
17
17
 
18
18
  import * as fs from "node:fs";
19
19
  import * as path from "node:path";
20
- import { randomUUID } from "node:crypto";
20
+ import { createHash, randomUUID } from "node:crypto";
21
+ import { execSync } from "node:child_process";
21
22
  import type {
22
23
  CompactionResult,
23
24
  ExtensionAPI,
@@ -44,7 +45,7 @@ import {
44
45
  type VccCompactionBuildResult,
45
46
  type VccCompactionStats,
46
47
  } from "./compaction.ts";
47
- import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
48
+ import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, runDirPath, setRunStatus, StateError, utcNow } from "./state.ts";
48
49
  import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
49
50
  import { bindRun, resolveActiveRun } from "./run-context.ts";
50
51
  import type { SubagentProgressEvent } from "./subagent.ts";
@@ -82,6 +83,7 @@ import {
82
83
  buildTaskView,
83
84
  currentTask,
84
85
  flattenTaskViews,
86
+ invalidateChecksForRolledBackTasks,
85
87
  taskIsTerminal,
86
88
  taskProgress,
87
89
  taskProgressMap,
@@ -97,8 +99,13 @@ import {
97
99
  renderDashboardLines,
98
100
  renderDashboardTreeLines,
99
101
  } from "./dashboard.ts";
100
- import { AUDIT_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit } from "./auditor.ts";
102
+ import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditRoundResult } from "./auditor.ts";
103
+ import { staleReloadHint as probeStaleReload } from "./staleness.ts";
101
104
  import { messaging } from "./messaging.ts";
105
+ import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
106
+ import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
107
+ import { resolveReviewerSpawn } from "./thinking-levels.ts";
108
+ import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
102
109
 
103
110
  export interface ExecState {
104
111
  planPath: string;
@@ -117,8 +124,14 @@ export interface ExecState {
117
124
  /** Stall watchdog (v0.6.1): consecutive settled rounds without a task
118
125
  * status change; auto-pause at the cap. */
119
126
  stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
120
- /** Completion-audit bookkeeping. */
121
- audit: { rounds: number; failed: string[]; running: boolean };
127
+ /** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
128
+ * only when a round outcome commits (never on discard/cancel) — and is the
129
+ * only piece persisted. */
130
+ audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
131
+ /** Execution-review loop (v0.8), memory-only: the attempt index names the
132
+ * per-round report files; consecutiveDiscards bounds the fingerprint
133
+ * re-run loop; inFlight owns the round's abort lifecycle. */
134
+ review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
122
135
  /** Per-settle audit latch (v0.7.1): a settled round fires the completion
123
136
  * audit at most once, so the turn_end / agent_before_settle / resume entry
124
137
  * points cannot double-consume a round when several land in one settle.
@@ -126,6 +139,18 @@ export interface ExecState {
126
139
  auditLatch?: { auditedThisSettle: boolean; activity: number };
127
140
  }
128
141
 
142
+ /** One in-flight review round: owns its abort lifecycle, its fingerprint of
143
+ * the audited subject, and the per-round one-shot wake token (v0.8). */
144
+ interface InFlightReview {
145
+ controller: AbortController;
146
+ /** Budget round this attempt belongs to (audit.rounds + 1 at spawn). */
147
+ budgetRound: number;
148
+ /** Monotonic attempt ordinal; names the round report file. */
149
+ attempt: number;
150
+ fingerprint: string;
151
+ wakeSent: boolean;
152
+ }
153
+
129
154
  /**
130
155
  * D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
131
156
  * and status bar right after a `set-language` change.
@@ -253,9 +278,12 @@ export function loadExecutionFromCheckpoint(
253
278
  const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
254
279
  const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
255
280
  const tasks = buildTaskView(planTasks, progress);
256
- // An audit-cap pause grants a fresh audit budget on restore (mirrors
257
- // resumeGoalWaitIfPaused) so the first turn can actually re-audit.
258
- const wasAuditCapPause = (cp.execution.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
281
+ // v0.8: a review-cap pause SURVIVES the restore — the budget must stay
282
+ // bounded across restarts; only /plans-execute grants a fresh one. The
283
+ // legacy v0.7 prefix stays dual-matched for one release.
284
+ const wasReviewCapPause = isReviewCapPause(cp.execution.pausedReason);
285
+ // Any live round from the replaced session graph dies here (CF2-002).
286
+ abortInFlightReview();
259
287
  if (!reverifyAll) {
260
288
  for (const id of cp.execution.doneVcIds) {
261
289
  const item = items.find((candidate) => candidate.id === id);
@@ -271,22 +299,24 @@ export function loadExecutionFromCheckpoint(
271
299
  startedAt: utcNow(),
272
300
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
273
301
  uiLanguage: resolveUiLanguage(ctx.cwd),
274
- // D-020: a paused legacy (or stopped) execution rebuilds unpaused —
275
- // the resume itself is the user's intent; the reason is surfaced in
276
- // the resume brief instead. An audit-cap pause additionally grants a
277
- // fresh audit budget here (mirrors resumeGoalWaitIfPaused), so the
278
- // first turn after a cross-session resume can actually re-audit.
279
- stall: {
280
- rounds: 0,
302
+ // D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
303
+ // resume itself is the user's intent; the reason is surfaced in the
304
+ // resume brief instead. EXCEPT a review-cap pause (v0.8): it must
305
+ // survive restores paused and at its committed round count, or the
306
+ // 5-round budget would never bound anything across restarts.
307
+ stall: {
308
+ rounds: cp.execution.stallRounds ?? 0,
281
309
  lastSnapshot: null,
282
- paused: false,
283
- pausedReason: undefined,
310
+ paused: wasReviewCapPause,
311
+ pausedReason: wasReviewCapPause ? (cp.execution.pausedReason ?? undefined) : undefined,
284
312
  },
285
313
  audit: {
286
- rounds: wasAuditCapPause ? 0 : (cp.execution.audit?.rounds ?? 0),
314
+ rounds: cp.execution.audit?.rounds ?? 0,
287
315
  failed: [],
316
+ undeterminable: cp.execution.audit?.undeterminable ?? [],
288
317
  running: false,
289
318
  },
319
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
290
320
  auditLatch: { auditedThisSettle: false, activity: 0 },
291
321
  };
292
322
  execution.stall.lastSnapshot = stallSnapshot();
@@ -295,11 +325,6 @@ export function loadExecutionFromCheckpoint(
295
325
  resetContinuationRuntime(ctx);
296
326
  pendingExecutionFlush = false;
297
327
  resetExecutionCompactionState(ctx);
298
- if (wasAuditCapPause) {
299
- withExecutionCheckpoint(ctx, (cp2) =>
300
- applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
301
- );
302
- }
303
328
  // D-020: an orphaned v0.6.0 delegated executor never survives a restart.
304
329
  // Its checkpoint delegate marker REFUSES the direct load — the run must
305
330
  // re-enter through the execution handoff so the C-006 approval gate
@@ -475,6 +500,8 @@ function updatePanelWidget(ctx: ExtensionContext): void {
475
500
  pausedReason: current.stall.pausedReason,
476
501
  auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
477
502
  auditFailed: current.audit.failed,
503
+ auditUndeterminable: current.audit.undeterminable,
504
+ reviewRunning: current.audit.running === true || current.review.inFlight !== null,
478
505
  startedAt: current.startedAt,
479
506
  usage: current.usage,
480
507
  });
@@ -497,6 +524,8 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
497
524
  pausedReason: execution.stall.pausedReason,
498
525
  auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
499
526
  auditFailed: execution.audit.failed,
527
+ auditUndeterminable: execution.audit.undeterminable,
528
+ reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
500
529
  });
501
530
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
502
531
  return;
@@ -524,6 +553,10 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
524
553
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
525
554
  return;
526
555
  }
556
+ if (status === "verifying") {
557
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `🔎 plans: ${active.run_id} (verifying)`));
558
+ return;
559
+ }
527
560
  if (status === "planning") {
528
561
  const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
529
562
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
@@ -594,6 +627,8 @@ export async function startExecution(
594
627
  ctx: ExtensionContext,
595
628
  input: StartExecutionInput,
596
629
  ): Promise<void> {
630
+ // A fresh handoff replaces any live run — abort its in-flight review round first.
631
+ abortInFlightReview();
597
632
  const tasks = buildTaskView(input.planTasks);
598
633
  execution = {
599
634
  planPath: input.planPath,
@@ -605,7 +640,8 @@ export async function startExecution(
605
640
  usage: { inToks: 0, outToks: 0 },
606
641
  uiLanguage: resolveUiLanguage(ctx.cwd),
607
642
  stall: { rounds: 0, lastSnapshot: null, paused: false },
608
- audit: { rounds: 0, failed: [], running: false },
643
+ audit: { rounds: 0, failed: [], undeterminable: [], running: false },
644
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
609
645
  auditLatch: { auditedThisSettle: false, activity: 0 },
610
646
  };
611
647
  // Seed the watchdog baseline only after `execution` points at the new state
@@ -670,9 +706,11 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
670
706
  applyExecutionProgress(cp, {
671
707
  tasks: taskProgressMap(execution!.tasks),
672
708
  doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
709
+ stallRounds: execution!.stall.rounds,
673
710
  audit: {
674
711
  rounds: execution!.audit.rounds,
675
712
  lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
713
+ undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
676
714
  },
677
715
  }),
678
716
  );
@@ -746,15 +784,17 @@ export function registerExecutionTurnHandlers(
746
784
  maybeContinuationFollowUp(ctx);
747
785
  return;
748
786
  }
749
- // Fully settled and still owed an audit: run it now. The audit's own
750
- // pass/fail message drives the rest (pass completes the run; fail
751
- // triggers a fix turn), so no extra continuation is requested here —
752
- // pendingAudit() is false once paused, which bounds any loop.
787
+ // Fully settled and still owed a review: launch the round under the
788
+ // mode rule — tui/rpc detach (the settle returns NOW and the overlay
789
+ // carries progress); print/json await inline so runtime teardown cannot
790
+ // kill the child. The round's own outcome routing drives the rest.
753
791
  latchAuditThisSettle();
754
- await runAuditFlow(ctx);
792
+ const chain = launchReviewRound(ctx);
793
+ if (chain) await chain;
755
794
  });
756
795
  ext.on("session_shutdown", async (_event, ctx) => {
757
796
  drainExecutionFlush(ctx);
797
+ abortInFlightReview();
758
798
  execution = null;
759
799
  executionRunId = null;
760
800
  continuationRuntime = null;
@@ -795,20 +835,205 @@ export function registerExecutionTurnHandlers(
795
835
  if (usage) recordExecutionTurn(ctx, usage);
796
836
  if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
797
837
  latchAuditThisSettle();
798
- await runAuditFlow(ctx);
838
+ const chain = launchReviewRound(ctx);
839
+ if (chain) await chain;
799
840
  }
800
841
  await onTurnEnd?.(ctx);
801
842
  });
802
843
  }
803
844
 
804
- /** Audit flow: run the completion auditor, apply pass/rollback, and either
805
- * complete the run, keep iterating (rollback), pause at the round cap
806
- * (interactive), or stop at the cap (auto-approve/headless — D-022).
845
+ /** ==== Execution-review loop (v0.8) ====
807
846
  *
808
- * Fail-closed completion: a run completes only when EVERY pending check was
809
- * affirmatively passed (or resolved skipped-pass); checks the auditor failed
810
- * to report count as failed, never as silently passed. */
811
- async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
847
+ * When every task is terminal and checks are still owed, the run enters the
848
+ * `verifying` status and a DETACHED read-only reviewer round runs in the
849
+ * background (tui/rpc): the settle handler returns immediately and the
850
+ * executor is truly idle while the overlay shows live progress. In print/json
851
+ * modes the settle handler keeps AWAITING the round inline — runtime
852
+ * teardown at settle would otherwise kill a detached child and swallow the
853
+ * pause signal.
854
+ *
855
+ * Budget: `audit.rounds` counts COMMITTED rounds only (pass, fail, or
856
+ * undeterminable); discards and cancellations burn nothing. Undeterminable
857
+ * rounds self-schedule the retry inside the loop (no wake). Two consecutive
858
+ * fingerprint discards commit as an undeterminable round so the loop stays
859
+ * bounded. Exhaustion pauses in EVERY mode (fail-closed) with an in-band
860
+ * `pi-plans-review-paused` message; the ONLY fresh-budget surface is
861
+ * /plans-execute — ordinary input and session restores never refill.
862
+ *
863
+ * Lifecycle: each round owns a session-scoped AbortController (never
864
+ * ctx.signal, which is turn-scoped), aborted from session_shutdown,
865
+ * stopExecution, startExecution, and restoreFromSession. Outcomes are
866
+ * guarded by execution identity (`execution !== owner` → silent discard).
867
+ *
868
+ * Completion stays fail-closed AND never fail-open: a run completes only
869
+ * when every pending check was affirmatively passed. */
870
+ const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
871
+ /** v0.7 protocol value — dual-matched for one release so checkpoints written
872
+ * by older builds keep their cap pause recognized on restore. */
873
+ const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
874
+
875
+ function isReviewCapPause(reason: string | undefined | null): boolean {
876
+ if (!reason) return false;
877
+ return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
878
+ }
879
+
880
+ /** The currently-running (or self-scheduling) review chain; the sanctioned
881
+ * test seam awaits this. */
882
+ let activeReviewChain: Promise<void> | null = null;
883
+
884
+ function abortInFlightReview(): void {
885
+ const inFlight = execution?.review.inFlight;
886
+ if (inFlight) {
887
+ try {
888
+ inFlight.controller.abort();
889
+ } catch {
890
+ /* already aborted */
891
+ }
892
+ if (execution) execution.review.inFlight = null;
893
+ }
894
+ activeReviewChain = null;
895
+ }
896
+
897
+ function reviewOwed(ex: ExecState): boolean {
898
+ return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
899
+ }
900
+
901
+ function runDirOf(ctx: ExtensionContext): string | null {
902
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
903
+ return runId ? runDirPath(ctx.cwd, runId) : null;
904
+ }
905
+
906
+ function setRunStatusForReview(ctx: ExtensionContext, status: "verifying" | "executing"): void {
907
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
908
+ if (!runId) return;
909
+ try {
910
+ const current = getRun(ctx.cwd, runId)?.status;
911
+ if (current !== status && current !== "done" && current !== "abandoned") {
912
+ setRunStatus(ctx.cwd, runId, status);
913
+ }
914
+ } catch {
915
+ /* best-effort */
916
+ }
917
+ }
918
+
919
+ /** Round fingerprint (Q-fingerprint-scope): plan digest + git HEAD +
920
+ * covered-file mtimes. A change between round start and resolve means the
921
+ * reviewer judged a subject that no longer exists — discard and re-run. */
922
+ function captureReviewFingerprint(ctx: ExtensionContext, ex: ExecState): string {
923
+ const parts: string[] = [];
924
+ try {
925
+ parts.push(createHash("sha256").update(fs.readFileSync(ex.planPath, "utf8")).digest("hex"));
926
+ } catch {
927
+ parts.push("plan-unreadable");
928
+ }
929
+ try {
930
+ parts.push(execSync("git rev-parse HEAD", { cwd: ctx.cwd, stdio: ["ignore", "pipe", "pipe"] }).toString().trim());
931
+ } catch {
932
+ parts.push("no-head");
933
+ }
934
+ const covered = new Set<string>();
935
+ for (const task of flattenTaskViews(ex.tasks)) {
936
+ for (const file of task.files ?? []) covered.add(file);
937
+ }
938
+ const mtimes: string[] = [];
939
+ for (const file of [...covered].sort()) {
940
+ try {
941
+ mtimes.push(`${file}:${fs.statSync(path.resolve(ctx.cwd, file)).mtimeMs}`);
942
+ } catch {
943
+ mtimes.push(`${file}:missing`);
944
+ }
945
+ }
946
+ parts.push(mtimes.join("|"));
947
+ return createHash("sha256").update(parts.join("\u0000")).digest("hex");
948
+ }
949
+
950
+ /** Round timeout: a committed round is minutes, never the 60-min subagent
951
+ * default — a hung child must surface as a spawn-failure round, not park the
952
+ * run in verifying for an hour (CF2-010). */
953
+ const REVIEW_ROUND_TIMEOUT_MS = 20 * 60 * 1000;
954
+
955
+ /** Reviewer-role pinning (Q-role-fallback): a CONFIRMED delegated role pins
956
+ * the spawn's model+thinking and labels the overlay with the role; an
957
+ * unconfirmed or current-session role inherits the session default with the
958
+ * "session default" label — a detached round NEVER opens the interactive
959
+ * first-use panel. */
960
+ function reviewSpawnProfile(): { model?: string; thinkingLevel?: string; label: string } {
961
+ try {
962
+ const reviewer = loadGlobalConfig().config.reviewer;
963
+ if (reviewer.mode !== "current-session" && reviewerReady(reviewer)) {
964
+ const spawn = resolveReviewerSpawn(reviewer);
965
+ if (spawn.modelSelector) {
966
+ return { model: spawn.modelSelector, thinkingLevel: spawn.thinkingLevel ?? undefined, label: spawn.label };
967
+ }
968
+ }
969
+ } catch {
970
+ /* fall through to the session default */
971
+ }
972
+ return { label: "session default" };
973
+ }
974
+
975
+ /** Engine-held lane state for the in-flight round: the reopen path builds a
976
+ * fresh one-shot controller seeded from THIS object, so the accumulated
977
+ * transcript survives ESC + reopen (CF2-001 / Q-reopen-seed). */
978
+ let reviewLane: RefineLaneState | null = null;
979
+ let reviewOverlay: RefineOverlayController | null = null;
980
+ let reviewModelLabel: string | undefined;
981
+
982
+ function freshReviewLane(attempt: number): RefineLaneState {
983
+ return {
984
+ id: `review-round-${attempt}`,
985
+ label: `Execution review round ${attempt}`,
986
+ status: "queued",
987
+ phase: "queued",
988
+ detail: "",
989
+ transcript: [],
990
+ currentTurnIndex: 0,
991
+ scrollOffset: 0,
992
+ followTranscript: true,
993
+ viewportHeight: 1,
994
+ };
995
+ }
996
+
997
+ /** Fresh controller per round (the controller is one-shot: closed latch,
998
+ * overlayPromise bail, terminal-lane early return — reuse drops progress).
999
+ * A UI failure must NEVER kill the round itself — best-effort only. */
1000
+ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
1001
+ if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
1002
+ try {
1003
+ const controller = new RefineOverlayController("auditor", [{ id: lane.id, label: lane.label }], () => {}, lang ?? "en");
1004
+ controller.seedLane(lane);
1005
+ controller.open(refineOverlayContext(ctx), modelLabel);
1006
+ return controller;
1007
+ } catch {
1008
+ return null;
1009
+ }
1010
+ }
1011
+
1012
+ /** The reopen surface (Task-3.4): rebuilds the overlay from engine-held lane
1013
+ * state; inert when no round is in flight. */
1014
+ export function reopenReviewOverlay(ctx: ExtensionContext): void {
1015
+ if (!execution?.review.inFlight || !reviewLane) return;
1016
+ const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
1017
+ if (controller) reviewOverlay = controller;
1018
+ }
1019
+
1020
+ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1021
+ const detail =
1022
+ ex.audit.failed.length > 0
1023
+ ? `failed: ${ex.audit.failed.join(", ")}`
1024
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1025
+ const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1026
+ pauseForStall(ctx, reason);
1027
+ // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
1028
+ // pauseForStall's ui.notify is optional and absent headless, so the pause
1029
+ // must also land in the session stream every mode can read.
1030
+ messaging().sendMessage(
1031
+ { customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
1032
+ { triggerTurn: false },
1033
+ );
1034
+ }
1035
+
1036
+ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
812
1037
  if (!execution) return;
813
1038
  const ex = execution;
814
1039
  // Skipped-pass checks resolve without a subagent round.
@@ -823,43 +1048,194 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
823
1048
  await completeExecution(ctx);
824
1049
  return;
825
1050
  }
826
- if (ex.audit.rounds >= AUDIT_MAX_ROUNDS) {
827
- // D-022: interactive sessions pause for the user (state kept, tasks
828
- // intact, resumable); auto-approve/headless terminates bounded.
829
- if (isInteractiveSession(ctx)) {
830
- pauseForStall(
831
- ctx,
832
- `${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
833
- );
834
- return;
835
- }
836
- await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
1051
+ if (ex.stall.paused || ex.review.inFlight) return;
1052
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1053
+ pauseReviewCap(ctx, ex);
837
1054
  return;
838
1055
  }
839
- ex.audit.running = true;
840
- ex.audit.rounds += 1;
1056
+ // Phase transition: the executor is done with its tasks; the review loop
1057
+ // owns the run until it converges (or pauses at the cap).
1058
+ setRunStatusForReview(ctx, "verifying");
1059
+ const round: InFlightReview = {
1060
+ controller: new AbortController(),
1061
+ budgetRound: ex.audit.rounds + 1,
1062
+ attempt: ex.review.attempts + 1,
1063
+ fingerprint: captureReviewFingerprint(ctx, ex),
1064
+ wakeSent: false,
1065
+ };
1066
+ ex.review.attempts = round.attempt;
1067
+ ex.review.inFlight = round;
1068
+ ex.audit.running = true; // dashboard mirror (the v0.8 model lands with the overlay task)
1069
+ const spawn = reviewSpawnProfile();
1070
+ reviewModelLabel = spawn.label;
1071
+ const lane = freshReviewLane(round.attempt);
1072
+ reviewLane = lane;
1073
+ reviewOverlay = openReviewOverlay(ctx, lane, ex.uiLanguage, spawn.label);
841
1074
  updateStatusWidget(ctx);
842
- let outcome = null as Awaited<ReturnType<typeof runCompletionAudit>>;
1075
+ let result: AuditRoundResult;
843
1076
  try {
844
- outcome = auditRunnerForTests
845
- ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: ex.audit.rounds })
1077
+ result = auditRunnerForTests
1078
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
846
1079
  : await runCompletionAudit(ctx, {
847
1080
  planPath: ex.planPath,
848
1081
  checklist: ex.items,
849
1082
  tasks: ex.tasks,
850
- round: ex.audit.rounds,
851
- signal: ctx.signal,
1083
+ round: round.attempt,
1084
+ model: spawn.model,
1085
+ thinkingLevel: spawn.thinkingLevel,
1086
+ timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
1087
+ signal: round.controller.signal,
1088
+ onProgress: (event) => {
1089
+ // The engine owns the lane state; the live controller only repaints.
1090
+ applyRefineProgress(lane, event);
1091
+ reviewOverlay?.rerender();
1092
+ },
1093
+ });
1094
+ } catch (error) {
1095
+ if (ex.review.inFlight === round) {
1096
+ ex.review.inFlight = null;
1097
+ ex.audit.running = false;
1098
+ }
1099
+ void reviewOverlay?.close();
1100
+ reviewOverlay = null;
1101
+ messaging().sendMessage(
1102
+ { customType: "pi-plans-review-error", content: `pi-plans: execution review round threw: ${String(error)}`, display: true },
1103
+ { triggerTurn: false },
1104
+ );
1105
+ updateStatusWidget(ctx);
1106
+ return;
1107
+ }
1108
+ // Overlay terminal state + close (the controller is one-shot; the engine-held
1109
+ // lane keeps the transcript for a later reopen within this round).
1110
+ const cancelledResult = result !== null && typeof result === "object" && "cancelled" in result;
1111
+ try {
1112
+ applyRefineResult(lane, {
1113
+ ok: !cancelledResult,
1114
+ output: result && "report" in result ? result.report : "",
1115
+ stderr: "",
1116
+ turns: 0,
1117
+ ...(cancelledResult ? { cancelled: true as const } : {}),
1118
+ });
1119
+ } catch {
1120
+ /* cosmetic only */
1121
+ }
1122
+ void reviewOverlay?.close();
1123
+ reviewOverlay = null;
1124
+ await handleReviewOutcome(ctx, ex, round, result, pendingChecks.map((item) => item.id));
1125
+ }
1126
+
1127
+ async function handleReviewOutcome(
1128
+ ctx: ExtensionContext,
1129
+ owner: ExecState,
1130
+ round: InFlightReview,
1131
+ result: AuditRoundResult,
1132
+ pendingIds: string[],
1133
+ ): Promise<void> {
1134
+ // Identity guard: a restore, stop, or fresh handoff replaced the run —
1135
+ // drop this outcome silently (CF2-002).
1136
+ if (execution !== owner) return;
1137
+ if (owner.review.inFlight !== round) return;
1138
+ owner.review.inFlight = null;
1139
+ owner.audit.running = false;
1140
+ if (result !== null && typeof result === "object" && "cancelled" in result) {
1141
+ // Aborted by shutdown/stop/restore/tree-switch: no round, no budget, no wake.
1142
+ updateStatusWidget(ctx);
1143
+ return;
1144
+ }
1145
+ const outcome = result;
1146
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1147
+ const fingerprintNow = captureReviewFingerprint(ctx, owner);
1148
+ const coveredTaskIds = flattenTaskViews(owner.tasks).map((task) => task.id);
1149
+ if (fingerprintNow !== round.fingerprint) {
1150
+ // The audited subject moved under the reviewer (Q-C): discard, re-run
1151
+ // without budget burn; two consecutive discards commit as undeterminable
1152
+ // so a mutating user cannot loop the loop for free (Q-discard-bound).
1153
+ owner.review.consecutiveDiscards += 1;
1154
+ const runDir = runDirOf(ctx);
1155
+ if (runDir) {
1156
+ writeReviewRoundReport(runDir, {
1157
+ budgetRound: round.budgetRound,
1158
+ attempt: round.attempt,
1159
+ outcome: "discarded",
1160
+ passed: [],
1161
+ failed: [],
1162
+ undeterminable: pendingIds,
1163
+ discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
1164
+ fingerprintCaptured: round.fingerprint,
1165
+ fingerprintFound: fingerprintNow,
1166
+ coveredTaskIds,
1167
+ report: reportText,
852
1168
  });
853
- } finally {
854
- ex.audit.running = false;
855
- }
856
- // Fail-closed: checks the outcome did not affirmatively pass are failed.
857
- const reportedPass = new Set(outcome?.passed ?? []);
858
- const failed = pendingChecks
859
- .map((item) => item.id)
860
- .filter((id) => !reportedPass.has(id));
861
- if (failed.length === 0) {
1169
+ }
1170
+ if (owner.review.consecutiveDiscards >= 2) {
1171
+ owner.review.consecutiveDiscards = 0;
1172
+ await commitReviewOutcome(
1173
+ ctx,
1174
+ owner,
1175
+ round,
1176
+ { passed: [], failed: [], undeterminable: pendingIds, report: `(two consecutive fingerprint discards — committed as an undeterminable round)\n\n${reportText}` },
1177
+ pendingIds,
1178
+ );
1179
+ return;
1180
+ }
1181
+ persist(ctx);
1182
+ updateStatusWidget(ctx);
1183
+ await maybeContinueReview(ctx, owner);
1184
+ return;
1185
+ }
1186
+ owner.review.consecutiveDiscards = 0;
1187
+ await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
1188
+ }
1189
+
1190
+ async function commitReviewOutcome(
1191
+ ctx: ExtensionContext,
1192
+ ex: ExecState,
1193
+ round: InFlightReview,
1194
+ outcome: { passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null,
1195
+ pendingIds: string[],
1196
+ ): Promise<void> {
1197
+ // The budget is charged only when an outcome commits — never on discard
1198
+ // or cancellation (CF2-003).
1199
+ ex.audit.rounds = round.budgetRound;
1200
+ const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
1201
+ const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
1202
+ // Anything the round neither passed nor failed is undeterminable: the
1203
+ // report omitted the check, spelled the verdict unreadably, or the
1204
+ // subagent never ran.
1205
+ const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
1206
+ const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
1207
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1208
+ {
1209
+ const runDir = runDirOf(ctx);
1210
+ if (runDir) {
1211
+ writeReviewRoundReport(runDir, {
1212
+ budgetRound: round.budgetRound,
1213
+ attempt: round.attempt,
1214
+ outcome: outcome === null
1215
+ ? "spawn-failed"
1216
+ : failed.length > 0
1217
+ ? "failed"
1218
+ : undeterminable.length > 0 ? "undeterminable" : "passed",
1219
+ passed,
1220
+ failed,
1221
+ undeterminable,
1222
+ fingerprintCaptured: round.fingerprint,
1223
+ coveredTaskIds,
1224
+ report: reportText,
1225
+ });
1226
+ }
1227
+ }
1228
+
1229
+ // Fail-closed completion: every pending check affirmatively passed. An
1230
+ // all-undeterminable round yields failed === [] — completing here would be
1231
+ // fail-open, marking a run done with nothing verified.
1232
+ if (passed.length === pendingIds.length) {
862
1233
  ex.audit.failed = [];
1234
+ ex.audit.undeterminable = [];
1235
+ for (const id of passed) {
1236
+ const item = ex.items.find((candidate) => candidate.id === id);
1237
+ if (item) item.done = true;
1238
+ }
863
1239
  withExecutionCheckpoint(ctx, (cp) =>
864
1240
  applyExecutionProgress(cp, {
865
1241
  tasks: taskProgressMap(ex.tasks),
@@ -870,18 +1246,21 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
870
1246
  await completeExecution(ctx);
871
1247
  return;
872
1248
  }
873
- // Failed checks: roll their covered tasks back (always — an unreported or
874
- // infra-failed audit must reopen work so the loop can continue) and
875
- // persist progress including checks that passed earlier rounds.
876
- for (const id of failed) {
1249
+ // Partial progress counts: a check affirmed this round is done even when a
1250
+ // sibling failed, so a later round only re-judges what is still open.
1251
+ for (const id of passed) {
877
1252
  const item = ex.items.find((candidate) => candidate.id === id);
878
- if (item) item.done = false;
1253
+ if (item) item.done = true;
879
1254
  }
880
1255
  ex.audit.failed = failed;
1256
+ ex.audit.undeterminable = undeterminable;
881
1257
  const rolledBack: string[] = [];
882
1258
  for (const id of failed) {
883
1259
  rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
884
1260
  }
1261
+ // A rollback reopens work other checks were verifying; those checks must
1262
+ // stop claiming the run is satisfied there.
1263
+ if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
885
1264
  withExecutionCheckpoint(ctx, (cp) =>
886
1265
  applyExecutionProgress(cp, {
887
1266
  tasks: taskProgressMap(ex.tasks),
@@ -889,29 +1268,95 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
889
1268
  audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
890
1269
  }),
891
1270
  );
892
- ex.stall.rounds = 0;
893
- ex.stall.lastSnapshot = stallSnapshot();
1271
+ if (rolledBack.length > 0) {
1272
+ // Rolling back is itself forward progress for the watchdog, but NOT for
1273
+ // audit.rounds: that counter stays monotonic so the repair loop is
1274
+ // bounded. The repair belongs to the executor — back to executing.
1275
+ ex.stall.rounds = 0;
1276
+ ex.stall.lastSnapshot = stallSnapshot();
1277
+ setRunStatusForReview(ctx, "executing");
1278
+ }
894
1279
  persist(ctx);
895
1280
  updateStatusWidget(ctx);
896
- const report = outcome?.report ?? "(audit subagent failed to run)";
897
- // v0.7.1: the failure notification now wakes the agent so it can fix the
898
- // rolled-back work without the user having to poke the run (root cause of
899
- // the observed stall). The wake is the ONLY turn this settle emits — the
900
- // rollback below also marks the runtime handled so the continuation path
901
- // cannot add a second EXECUTION_CONTINUE wake for the same settle (Q-2).
902
- const stranded = rolledBack.length === 0;
903
- messaging().sendMessage(
904
- {
905
- customType: "pi-plans-audit-failed",
906
- content: `**pi-plans: completion audit round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the audit reruns automatically once all tasks are terminal again.${ex.audit.rounds >= AUDIT_MAX_ROUNDS ? ` This was round ${AUDIT_MAX_ROUNDS} of ${AUDIT_MAX_ROUNDS}: interactive sessions pause for review; the next terminal-task cycle stops or pauses the run.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the audit automatically.` : ""}\n\n---\n${report.slice(0, 4000)}`,
907
- display: true,
908
- },
909
- { triggerTurn: true },
910
- );
911
- // Q-2 (single wake): this message owns the settle's only turn. Mark the
912
- // runtime handled so maybeContinuationFollowUp stays silent.
913
- const runtime = currentContinuationRuntime(ctx);
914
- if (runtime) runtime.handled = true;
1281
+ if (failed.length > 0) {
1282
+ // v0.8 wake: per-round one-shot token — exactly one triggerTurn per
1283
+ // committed failed outcome, even when the round resolves long after the
1284
+ // settle that spawned it. The continuation runtime is deliberately
1285
+ // untouched: detached rounds outlive their settle.
1286
+ if (!round.wakeSent) {
1287
+ round.wakeSent = true;
1288
+ const openTasks = flattenTaskViews(ex.tasks)
1289
+ .filter((task) => !taskIsTerminal(task))
1290
+ .map((task) => task.id);
1291
+ const stranded = rolledBack.length === 0;
1292
+ const content = `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the review automatically.` : ""}`;
1293
+ messaging().sendMessage(
1294
+ {
1295
+ customType: "pi-plans-audit-failed",
1296
+ content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}\n\n---\n${reportText.slice(0, 4000)}`,
1297
+ display: true,
1298
+ },
1299
+ { triggerTurn: true },
1300
+ );
1301
+ }
1302
+ return; // The agent repairs; the next settle re-enters the loop.
1303
+ }
1304
+ // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1305
+ // wake, no message; the dashboard/overlay carries the round counter.
1306
+ await maybeContinueReview(ctx, ex);
1307
+ }
1308
+
1309
+ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promise<void> {
1310
+ if (execution !== ex) return;
1311
+ if (!reviewOwed(ex)) return;
1312
+ if (ex.stall.paused) return;
1313
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1314
+ pauseReviewCap(ctx, ex);
1315
+ return;
1316
+ }
1317
+ await startReviewRound(ctx);
1318
+ }
1319
+
1320
+ /** Launch a review round under the mode rule: detached (fire-and-forget with
1321
+ * the chain tracked for the test seam) in tui/rpc; awaited inline otherwise
1322
+ * (print/json settle must hold the runtime open through the round). Returns
1323
+ * the chain when the caller must await it, null when detached. */
1324
+ function launchReviewRound(ctx: ExtensionContext): Promise<void> | null {
1325
+ const detach = ctx.mode === "tui" || ctx.mode === "rpc";
1326
+ const chain = (async () => {
1327
+ await startReviewRound(ctx);
1328
+ })();
1329
+ activeReviewChain = chain;
1330
+ if (detach) {
1331
+ chain.catch(() => {
1332
+ /* surfaced via the review messages */
1333
+ });
1334
+ return null;
1335
+ }
1336
+ return chain;
1337
+ }
1338
+
1339
+ /** Deterministic seam: await the in-flight (or self-scheduling) review chain. */
1340
+ export function __awaitReviewRoundForTests(): Promise<void> {
1341
+ return activeReviewChain ?? Promise.resolve();
1342
+ }
1343
+
1344
+ /** When this module graph was first imported into the running pi process.
1345
+ * pi loads extensions once, so a fix written to disk mid-session stays
1346
+ * invisible until /reload — which is exactly why an unreadable audit verdict
1347
+ * deserves a /reload hint rather than a bare retry. */
1348
+ const extensionModuleLoadedAt = new Date();
1349
+
1350
+ /** The /reload advice when the extension on disk is newer than the copy this
1351
+ * process loaded, else null. Shared with /plans via src/staleness.ts so both
1352
+ * report the same answer from one probe. */
1353
+ function staleReloadHint(): string | null {
1354
+ try {
1355
+ const root = path.dirname(path.dirname(new URL(import.meta.url).pathname));
1356
+ return probeStaleReload(root, extensionModuleLoadedAt);
1357
+ } catch {
1358
+ return null;
1359
+ }
915
1360
  }
916
1361
 
917
1362
  /** True when the session can surface a pause to a human (D-022): interactive
@@ -931,7 +1376,7 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
931
1376
  const run = getRun(ctx.cwd, active.run_id);
932
1377
  if (!run) return null;
933
1378
  if (phase === "planning" && run.status !== "planning") return null;
934
- if (phase === "execution" && run.status !== "executing") return null;
1379
+ if (phase === "execution" && run.status !== "executing" && run.status !== "verifying") return null;
935
1380
  scaffoldVccSettings(stateRoot);
936
1381
  return { settings: loadVccSettings(stateRoot), runId: run.run_id, artifactDir: run.artifact_dir };
937
1382
  }
@@ -1392,6 +1837,9 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
1392
1837
 
1393
1838
  export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
1394
1839
  if (!execution) return;
1840
+ // A stopped run's in-flight review round dies with it (typed cancelled —
1841
+ // no budget, no wake).
1842
+ abortInFlightReview();
1395
1843
  resetExecutionCompactionState(ctx);
1396
1844
  pendingExecutionFlush = false;
1397
1845
  persist(ctx);
@@ -1437,7 +1885,7 @@ function stallSnapshot(): string {
1437
1885
  }
1438
1886
 
1439
1887
  /**
1440
- * v0.7.1: shared "the completion audit is owed" predicate. Every entry point
1888
+ * v0.7.1: shared "the execution review is owed" predicate. Every entry point
1441
1889
  * that can start the audit (turn_end, agent_before_settle, restoreFromSession,
1442
1890
  * the resume path) goes through this so they can never disagree.
1443
1891
  *
@@ -1446,11 +1894,11 @@ function stallSnapshot(): string {
1446
1894
  */
1447
1895
  function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1448
1896
  if (!ex) return false;
1449
- // A paused run is never self-driven: the stall / audit-cap pause is an
1897
+ // A paused run is never self-driven: the stall / review-cap pause is an
1450
1898
  // explicit "hand control back" signal, and resuming it is the user's call.
1451
1899
  // This also bounds the zero-input continue loop in agent_before_settle.
1452
1900
  if (ex.stall.paused) return false;
1453
- if (ex.audit.running) return false;
1901
+ if (ex.review.inFlight) return false;
1454
1902
  return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
1455
1903
  }
1456
1904
 
@@ -1564,48 +2012,76 @@ export function filterContinuationMessages<T extends { customType?: string; deta
1564
2012
  return filterGoalWaitMessages(messages);
1565
2013
  }
1566
2014
 
1567
- /** Prefix of the stall reason used for the audit-cap pause (D-022). */
1568
- const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
1569
-
1570
2015
  /** Called for genuine user input or an explicit same-execution resume.
1571
- * Resuming an audit-cap pause grants a fresh audit budget (three more
1572
- * rounds): the user's explicit resume IS the decision to keep auditing —
1573
- * without this reset the cap pause could never be lifted productively. */
2016
+ * v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
2017
+ * refill the five-round budget (CF2-004); only /plans-execute
2018
+ * (resumeActiveExecution) is the explicit confirmation surface. Genuine
2019
+ * stall pauses still clear on input as before. */
1574
2020
  export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
1575
2021
  const ex = getExecution();
1576
2022
  if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
1577
- const wasAuditCap = (ex.stall.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
2023
+ if (isReviewCapPause(ex.stall.pausedReason)) {
2024
+ // Surfaced once per input so the user is not left guessing why the run
2025
+ // stays paused; the pause itself and the budget survive untouched.
2026
+ ctx.ui.notify?.(
2027
+ "pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
2028
+ "warning",
2029
+ );
2030
+ return false;
2031
+ }
1578
2032
  ex.stall.paused = false;
1579
2033
  ex.stall.pausedReason = undefined;
1580
2034
  ex.stall.rounds = 0;
1581
2035
  ex.stall.lastSnapshot = stallSnapshot();
1582
- if (wasAuditCap) {
1583
- ex.audit.rounds = 0;
1584
- ex.audit.failed = [];
2036
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
2037
+ persist(ctx);
2038
+ updateStatusWidget(ctx);
2039
+ return true;
2040
+ }
2041
+
2042
+ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2043
+ // v0.8: /plans-execute is THE explicit confirmation surface for a
2044
+ // review-cap pause — the only place a fresh five-round budget is granted
2045
+ // (Q-confirm-surface). Ordinary input and session restores never refill.
2046
+ const pausedEx = getExecution();
2047
+ if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
2048
+ pausedEx.stall.paused = false;
2049
+ pausedEx.stall.pausedReason = undefined;
2050
+ pausedEx.stall.rounds = 0;
2051
+ pausedEx.stall.lastSnapshot = stallSnapshot();
2052
+ pausedEx.audit.rounds = 0;
2053
+ pausedEx.audit.failed = [];
2054
+ pausedEx.audit.undeterminable = [];
1585
2055
  withExecutionCheckpoint(ctx, (cp) =>
1586
2056
  applyExecutionProgress(cp, {
1587
- tasks: taskProgressMap(ex.tasks),
2057
+ tasks: taskProgressMap(pausedEx.tasks),
1588
2058
  audit: { rounds: 0, lastResult: undefined },
1589
2059
  pausedReason: null,
1590
2060
  }),
1591
2061
  );
1592
- } else {
1593
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
2062
+ persist(ctx);
2063
+ updateStatusWidget(ctx);
2064
+ messaging().sendMessage(
2065
+ {
2066
+ customType: "pi-plans-review-budget-granted",
2067
+ content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
2068
+ display: true,
2069
+ },
2070
+ { triggerTurn: false },
2071
+ );
2072
+ const grantChain = launchReviewRound(ctx);
2073
+ if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
2074
+ return true;
1594
2075
  }
1595
- persist(ctx);
1596
- updateStatusWidget(ctx);
1597
- return true;
1598
- }
1599
-
1600
- export function resumeActiveExecution(ctx: ExtensionContext): boolean {
1601
2076
  // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
1602
2077
  // to `return false` here, so `/plans-execute` answered "already executing"
1603
2078
  // and the run stayed stranded until a full re-entry or a session restore.
1604
2079
  // It is not paused, so the pause path below cannot see it — check it first
1605
- // and run the owed audit instead of reporting "nothing to resume".
2080
+ // and run the owed review instead of reporting "nothing to resume".
1606
2081
  if (!execution?.stall.paused && pendingAudit()) {
1607
2082
  latchAuditThisSettle();
1608
- void runAuditFlow(ctx).catch(() => { /* surfaced via the audit message */ });
2083
+ const chain = launchReviewRound(ctx);
2084
+ if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
1609
2085
  return true;
1610
2086
  }
1611
2087
  if (!resumeGoalWaitIfPaused(ctx)) return false;
@@ -1635,7 +2111,7 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1635
2111
  messaging().sendMessage(
1636
2112
  {
1637
2113
  customType: "pi-plans-complete",
1638
- content: `**Plan complete!** ✅ \`${planPath}\` — completion audit passed.\n\n${summary}`,
2114
+ content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}`,
1639
2115
  display: true,
1640
2116
  },
1641
2117
  { triggerTurn: false },
@@ -1668,7 +2144,7 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
1668
2144
  ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
1669
2145
  : graphBlockForExecutor(mode === "enabled");
1670
2146
  const rollbackNote = execution.audit.failed.length > 0
1671
- ? `\nCompletion audit round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
2147
+ ? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
1672
2148
  : "";
1673
2149
  return `[PI-PLANS EXECUTION — write access enabled]
1674
2150
  Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
@@ -1684,7 +2160,7 @@ Execution rules:
1684
2160
  - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
1685
2161
  - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
1686
2162
  - Close subtasks before their parent; a parent is auditable only when every child is terminal.
1687
- - When every task is terminal, the independent completion auditor verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
2163
+ - When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
1688
2164
  - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1689
2165
  - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
1690
2166
  - Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
@@ -1745,6 +2221,9 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1745
2221
  } catch {
1746
2222
  tasks = snapshot.tasks;
1747
2223
  }
2224
+ // The snapshot cannot carry an in-flight round (it is memory-only); abort
2225
+ // any live one from the previous session graph and rebuild fresh (CF2-002).
2226
+ abortInFlightReview();
1748
2227
  execution = {
1749
2228
  planPath: snapshot.planPath,
1750
2229
  items,
@@ -1754,8 +2233,14 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1754
2233
  startedAt: snapshot.startedAt,
1755
2234
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
1756
2235
  uiLanguage: resolveUiLanguage(ctx.cwd),
1757
- stall: { ...snapshot.stall, lastSnapshot: null, rounds: 0 },
1758
- audit: { rounds: snapshot.audit?.rounds ?? 0, failed: snapshot.audit?.failed ?? [], running: false },
2236
+ stall: { ...snapshot.stall, lastSnapshot: null },
2237
+ audit: {
2238
+ rounds: snapshot.audit?.rounds ?? 0,
2239
+ failed: snapshot.audit?.failed ?? [],
2240
+ undeterminable: [],
2241
+ running: false,
2242
+ },
2243
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
1759
2244
  auditLatch: { auditedThisSettle: false, activity: 0 },
1760
2245
  };
1761
2246
  execution.stall.lastSnapshot = stallSnapshot();
@@ -1765,9 +2250,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
1765
2250
  if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1766
2251
  persist(ctx);
1767
2252
  if (pendingAudit()) {
1768
- // Terminal tasks without a passing audit: rerun the audit flow.
2253
+ // Self-heal (v0.8): a verifying run with pending checks restarts its
2254
+ // round exactly once per resume — the full predicate (paused, in-flight,
2255
+ // budget) lives inside startReviewRound. Restores never grant budget.
1769
2256
  latchAuditThisSettle();
1770
- await runAuditFlow(ctx);
2257
+ const chain = launchReviewRound(ctx);
2258
+ if (chain) await chain;
1771
2259
  }
1772
2260
  updateStatusWidget(ctx);
1773
2261
  }