pi-plans 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/exec.ts CHANGED
@@ -57,6 +57,7 @@ import {
57
57
  applyExecutionProgress,
58
58
  applyExecutionStopped,
59
59
  createCheckpoint,
60
+ applyExecutionPlanAmended,
60
61
  loadCheckpoint,
61
62
  mutateCheckpoint,
62
63
  planIdentityOf,
@@ -65,6 +66,8 @@ import {
65
66
  sha256File,
66
67
  StaleCheckpointError,
67
68
  type ExecutionApproval,
69
+ type ExecutionBlocked,
70
+ type ExecutionCheckpoint,
68
71
  type WorkflowCheckpoint,
69
72
  } from "./workflow-state.ts";
70
73
  import { graphBlockForExecutor } from "./code-graph/prompts.ts";
@@ -72,6 +75,7 @@ import { resolveGraphMode } from "./code-graph/mode.ts";
72
75
  import {
73
76
  parseChecklist,
74
77
  parsePlanTasks,
78
+ CHECKLIST_HEADERS,
75
79
  flattenTasks,
76
80
  type CheckItem,
77
81
  type PlanTasks,
@@ -80,10 +84,14 @@ import {
80
84
  allTasksTerminal,
81
85
  auditRollbackSet,
82
86
  auditableChecks,
87
+ blockedReviewTasks,
83
88
  buildTaskView,
84
89
  currentTask,
90
+ findingsRollbackSet,
85
91
  flattenTaskViews,
86
92
  invalidateChecksForRolledBackTasks,
93
+ maxWave,
94
+ rollbackCoverageIds,
87
95
  taskIsTerminal,
88
96
  taskProgress,
89
97
  taskProgressMap,
@@ -99,13 +107,35 @@ import {
99
107
  renderDashboardLines,
100
108
  renderDashboardTreeLines,
101
109
  } from "./dashboard.ts";
102
- import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditRoundResult } from "./auditor.ts";
110
+ import { presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
111
+ import {
112
+ DEFAULT_REVIEW_BUDGET,
113
+ LEGACY_REVIEW_MAX_ROUNDS,
114
+ NO_PROGRESS_MAX_STREAK,
115
+ REVIEW_CAP_PAUSE_PREFIX,
116
+ REVIEW_NO_PROGRESS_PAUSE_PREFIX,
117
+ UNLIMITED_HARD_CAP,
118
+ askReviewBudget,
119
+ budgetExhausted,
120
+ bumpNoProgress,
121
+ formatReviewBudget,
122
+ isReviewPauseReason,
123
+ noProgressSignature,
124
+ noProgressTripped,
125
+ resolveStoredBudget,
126
+ reviewBudgetPanelAvailable,
127
+ unlimitedHardCapCeiling,
128
+ type NoProgressState,
129
+ type ReviewBudget,
130
+ } from "./review-budget.ts";
131
+ import type { ReviewFindingRecord } from "./workflow-state.ts";
103
132
  import { staleReloadHint as probeStaleReload } from "./staleness.ts";
104
133
  import { messaging } from "./messaging.ts";
105
134
  import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
106
135
  import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
107
136
  import { resolveReviewerSpawn } from "./thinking-levels.ts";
108
137
  import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
138
+ import { matchesTerminalKey } from "./terminal-keys.ts";
109
139
 
110
140
  export interface ExecState {
111
141
  planPath: string;
@@ -127,17 +157,52 @@ export interface ExecState {
127
157
  /** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
128
158
  * only when a round outcome commits (never on discard/cancel) — and is the
129
159
  * only piece persisted. */
130
- audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
160
+ audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean; findings: ReviewFinding[] };
131
161
  /** Execution-review loop (v0.8), memory-only: the attempt index names the
132
162
  * per-round report files; consecutiveDiscards bounds the fingerprint
133
163
  * re-run loop; inFlight owns the round's abort lifecycle. */
134
- review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
164
+ review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null; budgetAsking?: boolean };
135
165
  /** Per-settle audit latch (v0.7.1): a settled round fires the completion
136
166
  * audit at most once, so the turn_end / agent_before_settle / resume entry
137
167
  * points cannot double-consume a round when several land in one settle.
138
168
  * Created on demand by auditLatchOf(); every construction path may omit it. */
139
169
  auditLatch?: { auditedThisSettle: boolean; activity: number };
140
- }
170
+ /** v0.9.2: outstanding blocker from the newest failed review round — the
171
+ * authoritative rollback set captured at commit time plus the still-open
172
+ * tasks among it. Memory mirror of `execution.blocked`; null = nothing
173
+ * blocks the review. */
174
+ blocked: ExecBlocked | null;
175
+ /** v0.9.2: blocker ids as observed at the PREVIOUS blocked wake — the
176
+ * ladder's progress baseline. Memory-only: `blocked.tasks` is re-synced by
177
+ * `persistTaskProgress` on every task close (so it stays fresh for the
178
+ * resume brief), which would make a comparison against it meaningless.
179
+ * Absent after a restore/resume → the next blocked wake restarts the count. */
180
+ blockedWakeTasks?: string[];
181
+ /** v0.9.2: highest escalation level already surfaced as a visible system
182
+ * line. Memory-only (never persisted/snapshotted) so a restored session
183
+ * re-notifies once. */
184
+ blockedNotifiedLevel?: number;
185
+ /** v0.9.3: the per-run execution-review budget. Resolved exactly once —
186
+ * right before round 1 — from the picker or the no-UI default, then
187
+ * persisted. `undefined` = not decided yet (never conflated with a legacy
188
+ * checkpoint, which `loadExecutionFromCheckpoint` resolves to 5). */
189
+ reviewBudget?: ReviewBudget;
190
+ /** v0.9.3: the budget came from the fallback (no UI/headless/auto-approve),
191
+ * not from a user pick — the expanded dashboard row marks it `(default)`
192
+ * and the resume brief repeats the note. Persisted so a later session can
193
+ * still tell the difference. */
194
+ reviewBudgetDefaulted?: boolean;
195
+ /** v0.9.3: committed review rounds across the whole run (never reset by a
196
+ * grant) — the unlimited budget's cumulative hard-cap counter. */
197
+ reviewRoundsTotal: number;
198
+ /** v0.9.3: rounds added to the unlimited hard cap by explicit grants. */
199
+ reviewCapExtension: number;
200
+ /** v0.9.3: no-progress valve state (unlimited budget only). */
201
+ reviewNoProgress?: NoProgressState;
202
+ }
203
+
204
+ /** Live blocker record (see `ExecutionBlocked` in workflow-state.ts). */
205
+ type ExecBlocked = ExecutionBlocked;
141
206
 
142
207
  /** One in-flight review round: owns its abort lifecycle, its fingerprint of
143
208
  * the audited subject, and the per-round one-shot wake token (v0.8). */
@@ -167,6 +232,8 @@ const STALL_MAX_ROUNDS = 3;
167
232
  let execution: ExecState | null = null;
168
233
 
169
234
  export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
235
+ /** v0.9.2: visible escalation line — never replays after a restart. */
236
+ export const EXECUTION_BLOCKED_CUSTOM_TYPE = "pi-plans-exec-blocked";
170
237
  /** Legacy v0.6.0 continuation message type — filtered on restore. */
171
238
  const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
172
239
 
@@ -229,9 +296,55 @@ export interface CheckpointExecutionLoad {
229
296
  /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
230
297
  * detected — resume requires a fresh handoff approval. */
231
298
  legacyDelegate?: boolean;
299
+ /** v0.9.1 (F-005): unresolved findings from the newest committed round,
300
+ * so the /resume-plans brief can surface outstanding highs before the
301
+ * per-turn injection ever runs. */
302
+ findings?: ReviewFinding[];
303
+ /** v0.9.2: persisted blocker of the newest failed round, so the resume
304
+ * brief can name the open tasks that keep the review from starting. */
305
+ blocked?: ExecutionBlocked | null;
306
+ /** v0.9.3: the persisted per-run review budget (or undefined when the run
307
+ * never picked one — a legacy checkpoint resolves to 5 in the live state,
308
+ * see `reviewBudgetDefaulted`). */
309
+ reviewBudget?: ReviewBudget;
310
+ reviewBudgetDefaulted?: boolean;
311
+ reviewRoundsTotal?: number;
312
+ reviewCapExtension?: number;
232
313
  error?: string;
233
314
  }
234
315
 
316
+ /** v0.9.2: a checkpoint written before `execution.blocked` existed still names
317
+ * its blocker — `audit.lastResult` records the newest failed round's ids and
318
+ * the coverage cascade is pure, so the rollback set can be reconstructed
319
+ * WITHOUT re-applying anything (the tree already carries the reopen's outcome;
320
+ * re-applying it would revert tasks the executor has since re-closed).
321
+ * Finding-driven rounds (`highs: F-…`) carry no coverage and stay without a
322
+ * record — the wake still states that no round can start while tasks are open. */
323
+ function blockedFromFailedIds(
324
+ failedIds: readonly string[],
325
+ round: number,
326
+ tasks: TaskView[],
327
+ checklist: CheckItem[],
328
+ ): ExecBlocked | null {
329
+ if (failedIds.length === 0) return null;
330
+ const rolledBack = rollbackCoverageIds(tasks, checklist, failedIds, []);
331
+ if (rolledBack.length === 0) return null;
332
+ const open = blockedReviewTasks(tasks, rolledBack);
333
+ if (open.length === 0) return null;
334
+ return { rolledBack, tasks: open, round, escalatedRounds: 0, since: utcNow() };
335
+ }
336
+
337
+ function backfillBlocked(
338
+ checkpoint: ExecutionCheckpoint,
339
+ tasks: TaskView[],
340
+ checklist: CheckItem[],
341
+ ): ExecBlocked | null {
342
+ const last = checkpoint.audit?.lastResult?.trim() ?? "";
343
+ if (last.length === 0 || last.startsWith("highs:")) return null;
344
+ const failed = last.split(",").map((id) => id.trim()).filter((id) => id.length > 0);
345
+ return blockedFromFailedIds(failed, checkpoint.audit?.rounds ?? 0, tasks, checklist);
346
+ }
347
+
235
348
  /**
236
349
  * Shared restore primitive: load the executing state from a run checkpoint
237
350
  * into THIS session. Authorization is kept only when the recorded approval
@@ -299,6 +412,11 @@ export function loadExecutionFromCheckpoint(
299
412
  startedAt: utcNow(),
300
413
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
301
414
  uiLanguage: resolveUiLanguage(ctx.cwd),
415
+ // v0.9.2: the persisted blocker survives a restore so the resume brief
416
+ // and the dashboard can name it; a reverifyAll restore drops it together
417
+ // with the task progress it described. Pre-feature checkpoints are
418
+ // backfilled from the newest failed round's ids.
419
+ blocked: reverifyAll ? null : (cp.execution.blocked ?? backfillBlocked(cp.execution, tasks, items)),
302
420
  // D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
303
421
  // resume itself is the user's intent; the reason is surfaced in the
304
422
  // resume brief instead. EXCEPT a review-cap pause (v0.8): it must
@@ -314,10 +432,19 @@ export function loadExecutionFromCheckpoint(
314
432
  rounds: cp.execution.audit?.rounds ?? 0,
315
433
  failed: [],
316
434
  undeterminable: cp.execution.audit?.undeterminable ?? [],
435
+ findings: toReviewFindings(cp.execution.audit?.findings),
317
436
  running: false,
318
437
  },
319
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
438
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
320
439
  auditLatch: { auditedThisSettle: false, activity: 0 },
440
+ // v0.9.3: the review budget survives a restore; a checkpoint written
441
+ // before the feature that already spent rounds keeps the legacy 5-round
442
+ // bound instead of being cut to the new default mid-flight.
443
+ reviewBudget: resolveStoredBudget(cp.execution.reviewBudget, cp.execution.audit?.rounds ?? 0),
444
+ reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted === true,
445
+ reviewRoundsTotal: cp.execution.reviewRoundsTotal ?? 0,
446
+ reviewCapExtension: cp.execution.reviewCapExtension ?? 0,
447
+ reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
321
448
  };
322
449
  execution.stall.lastSnapshot = stallSnapshot();
323
450
  executionRunId = runId;
@@ -352,6 +479,11 @@ export function loadExecutionFromCheckpoint(
352
479
  if (headChanged) {
353
480
  withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
354
481
  }
482
+ // A backfilled blocker is authoritative from here on: persist it once so a
483
+ // later restart reads the record instead of re-deriving it.
484
+ if (!reverifyAll && (cp.execution.blocked ?? null) === null && execution!.blocked !== null) {
485
+ withExecutionCheckpoint(ctx, (current) => applyExecutionProgress(current, { blocked: execution!.blocked }));
486
+ }
355
487
  persist(ctx);
356
488
  updateStatusWidget(ctx);
357
489
  return {
@@ -362,6 +494,14 @@ export function loadExecutionFromCheckpoint(
362
494
  pausedReason: cp.execution.pausedReason,
363
495
  legacyPlan: planTasks.legacy,
364
496
  legacyDelegate,
497
+ findings: toReviewFindings(cp.execution.audit?.findings),
498
+ // The LIVE record: a pre-feature checkpoint is backfilled (and persisted)
499
+ // above, so the resume brief sees the reconstructed blocker.
500
+ blocked: execution?.blocked ?? null,
501
+ reviewBudget: execution?.reviewBudget,
502
+ reviewBudgetDefaulted: execution?.reviewBudgetDefaulted === true,
503
+ reviewRoundsTotal: execution?.reviewRoundsTotal ?? 0,
504
+ reviewCapExtension: execution?.reviewCapExtension ?? 0,
365
505
  };
366
506
  }
367
507
 
@@ -501,7 +641,10 @@ function updatePanelWidget(ctx: ExtensionContext): void {
501
641
  auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
502
642
  auditFailed: current.audit.failed,
503
643
  auditUndeterminable: current.audit.undeterminable,
644
+ findings: current.audit.findings,
504
645
  reviewRunning: current.audit.running === true || current.review.inFlight !== null,
646
+ blockedTasks: blockedTaskIds(current),
647
+ blockedRound: current.blocked?.round ?? null,
505
648
  startedAt: current.startedAt,
506
649
  usage: current.usage,
507
650
  });
@@ -525,7 +668,14 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
525
668
  auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
526
669
  auditFailed: execution.audit.failed,
527
670
  auditUndeterminable: execution.audit.undeterminable,
671
+ findings: execution.audit.findings,
528
672
  reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
673
+ blockedTasks: blockedTaskIds(execution),
674
+ blockedRound: execution.blocked?.round ?? null,
675
+ reviewBudget: execution.reviewBudget ?? null,
676
+ reviewRoundsTotal: execution.reviewRoundsTotal,
677
+ reviewCapExtension: execution.reviewCapExtension,
678
+ reviewBudgetDefaulted: execution.reviewBudgetDefaulted === true,
529
679
  });
530
680
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
531
681
  return;
@@ -597,7 +747,16 @@ function persist(ctx: ExtensionContext): void {
597
747
  startedAt: execution.startedAt,
598
748
  usage: execution.usage,
599
749
  stall: execution.stall,
600
- audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
750
+ blocked: execution.blocked,
751
+ // v0.9.3 (round-1 F-005): the review budget rides the session snapshot
752
+ // too — a /reload restore rebuilds the live state from it, and a
753
+ // restored counter must NOT hand the run a free unlimited window.
754
+ reviewBudget: execution.reviewBudget,
755
+ reviewBudgetDefaulted: execution.reviewBudgetDefaulted,
756
+ reviewRoundsTotal: execution.reviewRoundsTotal,
757
+ reviewCapExtension: execution.reviewCapExtension,
758
+ reviewNoProgress: execution.reviewNoProgress,
759
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
601
760
  });
602
761
  }
603
762
 
@@ -640,9 +799,15 @@ export async function startExecution(
640
799
  usage: { inToks: 0, outToks: 0 },
641
800
  uiLanguage: resolveUiLanguage(ctx.cwd),
642
801
  stall: { rounds: 0, lastSnapshot: null, paused: false },
643
- audit: { rounds: 0, failed: [], undeterminable: [], running: false },
644
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
802
+ blocked: null,
803
+ audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
804
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
645
805
  auditLatch: { auditedThisSettle: false, activity: 0 },
806
+ // v0.9.3: a fresh handoff starts with NO budget — the picker resolves it
807
+ // (or the no-UI default applies) right before round 1.
808
+ reviewBudget: undefined,
809
+ reviewRoundsTotal: 0,
810
+ reviewCapExtension: 0,
646
811
  };
647
812
  // Seed the watchdog baseline only after `execution` points at the new state
648
813
  // (stallSnapshot reads the live execution).
@@ -702,15 +867,28 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
702
867
  // Any task-state change resets the stall watchdog baseline.
703
868
  execution.stall.rounds = 0;
704
869
  execution.stall.lastSnapshot = stallSnapshot();
870
+ // v0.9.2: closing a reopened task shrinks the blocker; when the last one
871
+ // closes the record is dropped so no stale blocker outlives the repair.
872
+ if (execution.blocked) {
873
+ execution.blocked.tasks = blockedReviewTasks(execution.tasks, execution.blocked.rolledBack);
874
+ if (execution.blocked.tasks.length === 0) {
875
+ execution.blocked = null;
876
+ execution.blockedWakeTasks = undefined;
877
+ }
878
+ }
705
879
  withExecutionCheckpoint(ctx, (cp) =>
706
880
  applyExecutionProgress(cp, {
707
881
  tasks: taskProgressMap(execution!.tasks),
708
882
  doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
709
883
  stallRounds: execution!.stall.rounds,
884
+ blocked: execution!.blocked,
710
885
  audit: {
711
886
  rounds: execution!.audit.rounds,
712
887
  lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
713
888
  undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
889
+ // v0.9: audit writes are replace-semantics — every writer must
890
+ // carry findings or a task update would silently wipe them.
891
+ findings: execution!.audit.findings,
714
892
  },
715
893
  }),
716
894
  );
@@ -737,10 +915,10 @@ export function recordExecutionTurn(
737
915
  }
738
916
 
739
917
  /** Test hook: replace the audit subagent with a deterministic function. */
740
- let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
918
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
741
919
 
742
920
  export function __setAuditRunnerForTests(
743
- runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
921
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
744
922
  ): void {
745
923
  auditRunnerForTests = runner;
746
924
  }
@@ -867,14 +1045,102 @@ export function registerExecutionTurnHandlers(
867
1045
  *
868
1046
  * Completion stays fail-closed AND never fail-open: a run completes only
869
1047
  * when every pending check was affirmatively passed. */
870
- const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
871
- /** v0.7 protocol value — dual-matched for one release so checkpoints written
872
- * by older builds keep their cap pause recognized on restore. */
873
- const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
874
-
875
1048
  function isReviewCapPause(reason: string | undefined | null): boolean {
876
- if (!reason) return false;
877
- return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
1049
+ return isReviewPauseReason(reason);
1050
+ }
1051
+
1052
+ /** v0.9.3: the counters that bound an unlimited budget. */
1053
+ function budgetCounters(ex: ExecState): { reviewRoundsTotal: number; reviewCapExtension: number } {
1054
+ return { reviewRoundsTotal: ex.reviewRoundsTotal, reviewCapExtension: ex.reviewCapExtension };
1055
+ }
1056
+
1057
+ /** The budget in force for this run; a legacy checkpoint resolves to 5 during
1058
+ * load, so this is only a guard for hand-built test states. */
1059
+ function activeBudget(ex: ExecState): ReviewBudget {
1060
+ return ex.reviewBudget ?? LEGACY_REVIEW_MAX_ROUNDS;
1061
+ }
1062
+
1063
+ function budgetSpent(ex: ExecState): boolean {
1064
+ return budgetExhausted(activeBudget(ex), ex.audit.rounds, budgetCounters(ex));
1065
+ }
1066
+
1067
+ /** `3` / `∞` — the budget as shown in status lines and messages. */
1068
+ function budgetLabel(ex: ExecState): string {
1069
+ return formatReviewBudget(activeBudget(ex));
1070
+ }
1071
+
1072
+ /**
1073
+ * v0.9.3: whether the budget panel may be shown in THIS session. Beyond the
1074
+ * native-surface check, auto-approve sessions never ask — the recorded plan
1075
+ * decision routes headless/auto-approve runs to the default budget
1076
+ * (`PI_PLANS_AUTO_APPROVE=1` exists for unattended harnesses, where a panel
1077
+ * would either hang or answer meaninglessly).
1078
+ */
1079
+ function reviewBudgetPanelUsable(ctx: ExtensionContext): boolean {
1080
+ return !isAutoApproveEnabledLocal() && reviewBudgetPanelAvailable(ctx);
1081
+ }
1082
+
1083
+ /** Persist a freshly resolved budget (and the counters) in one revision. */
1084
+ function persistResolvedBudget(ctx: ExtensionContext, ex: ExecState): void {
1085
+ withExecutionCheckpoint(ctx, (cp) =>
1086
+ applyExecutionProgress(cp, {
1087
+ reviewBudget: ex.reviewBudget ?? null,
1088
+ reviewBudgetDefaulted: ex.reviewBudgetDefaulted === true,
1089
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1090
+ reviewCapExtension: ex.reviewCapExtension,
1091
+ }),
1092
+ );
1093
+ persist(ctx);
1094
+ updateStatusWidget(ctx);
1095
+ }
1096
+
1097
+ /** One visible note whenever the fallback (rather than a user pick) decided
1098
+ * the budget — never a silent bound (v0.9.3, Q-3). */
1099
+ function notifyDefaultBudget(ex: ExecState, why: "no-panel" | "cancelled"): void {
1100
+ const budget = formatReviewBudget(ex.reviewBudget ?? DEFAULT_REVIEW_BUDGET);
1101
+ messaging().sendMessage(
1102
+ {
1103
+ customType: "pi-plans-review-budget-default",
1104
+ content:
1105
+ why === "no-panel"
1106
+ ? `**pi-plans: execution review budget: ${budget} (default)** — no budget panel is available in this session, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`
1107
+ : `**pi-plans: execution review budget: ${budget} (default)** — the budget panel was closed without a choice, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`,
1108
+ display: true,
1109
+ },
1110
+ { triggerTurn: false },
1111
+ );
1112
+ }
1113
+
1114
+ /** Synchronous fallback: apply (and persist) the default budget. Used when no
1115
+ * panel exists at all — the headless/print/json and auto-approve paths — so
1116
+ * the first round needs no extra async hop. */
1117
+ function applyDefaultReviewBudget(ctx: ExtensionContext, ex: ExecState): void {
1118
+ if (ex.reviewBudget !== undefined) return;
1119
+ ex.reviewBudget = DEFAULT_REVIEW_BUDGET;
1120
+ ex.reviewBudgetDefaulted = true;
1121
+ notifyDefaultBudget(ex, "no-panel");
1122
+ persistResolvedBudget(ctx, ex);
1123
+ }
1124
+
1125
+ /**
1126
+ * Resolve the per-run review budget exactly once, immediately before the first
1127
+ * round. The panel path is async; every other case is handled up front by
1128
+ * `applyDefaultReviewBudget`. `budgetAsking` keeps a second settle (or a
1129
+ * restore while the panel is open) from opening a second panel.
1130
+ */
1131
+ async function askReviewBudgetForRun(ctx: ExtensionContext, ex: ExecState): Promise<void> {
1132
+ if (ex.reviewBudget !== undefined || ex.review.budgetAsking) return;
1133
+ ex.review.budgetAsking = true;
1134
+ try {
1135
+ const picked = await askReviewBudget(ctx, ex.uiLanguage, undefined);
1136
+ if (execution !== ex || ex.reviewBudget !== undefined) return;
1137
+ ex.reviewBudget = picked ?? DEFAULT_REVIEW_BUDGET;
1138
+ ex.reviewBudgetDefaulted = picked === null;
1139
+ if (picked === null) notifyDefaultBudget(ex, "cancelled");
1140
+ persistResolvedBudget(ctx, ex);
1141
+ } finally {
1142
+ ex.review.budgetAsking = false;
1143
+ }
878
1144
  }
879
1145
 
880
1146
  /** The currently-running (or self-scheduling) review chain; the sanctioned
@@ -894,8 +1160,78 @@ function abortInFlightReview(): void {
894
1160
  activeReviewChain = null;
895
1161
  }
896
1162
 
1163
+ /** v0.9: unresolved high-severity findings from the newest committed round.
1164
+ * Presence in the newest round's report IS the unresolved set (stable ids:
1165
+ * a problem is resolved only by no longer being reported). */
1166
+ function unresolvedHighFindings(ex: ExecState): ReviewFinding[] {
1167
+ return ex.audit.findings.filter((f) => f.severity === "high");
1168
+ }
1169
+
1170
+ /** v0.9: checkpoint records -> runtime findings. Invalid severities degrade to
1171
+ * "malformed" (recorded, non-blocking) — the same never-fail posture as the
1172
+ * report parser. */
1173
+ function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
1174
+ if (!records) return [];
1175
+ return records.map((r) => {
1176
+ const severity = (r.severity === "high" || r.severity === "medium" || r.severity === "low") ? r.severity : "malformed";
1177
+ return { id: r.id, severity, taskIds: Array.isArray(r.taskIds) ? r.taskIds : [], proposedTask: r.proposedTask, note: r.note ?? "", evidence: r.evidence ?? "", raw: r.raw ?? "" };
1178
+ });
1179
+ }
1180
+
1181
+ /** v0.9: findings widen the owed predicate — an unresolved high finding keeps
1182
+ * the review owed even when every check is done (the stranded-high path),
1183
+ * mirroring how a failed check keeps it owed today. */
897
1184
  function reviewOwed(ex: ExecState): boolean {
898
- return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
1185
+ return allTasksTerminal(ex.tasks) && reviewOutstanding(ex);
1186
+ }
1187
+
1188
+ /** v0.9.2: the review is outstanding wherever the tree stands — checks still
1189
+ * owed or an unresolved high finding. `reviewOwed` ANDs this with
1190
+ * allTasksTerminal; the blocked wake needs exactly the half that stays true
1191
+ * while tasks are open, because that is the state in which the review cannot
1192
+ * start (and in which `pendingAudit` is false by construction). */
1193
+ function reviewOutstanding(ex: ExecState): boolean {
1194
+ return auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0;
1195
+ }
1196
+
1197
+ /** v0.9.2: live blocker ids — the newest failed round's rollback set
1198
+ * intersected with the still-open tasks. Always recomputed from the tree: the
1199
+ * stored `tasks` list may predate the last status change. */
1200
+ function blockedTaskIds(ex: ExecState): string[] {
1201
+ return ex.blocked ? blockedReviewTasks(ex.tasks, ex.blocked.rolledBack) : [];
1202
+ }
1203
+
1204
+ /** v0.9.2: one visible system line per escalation level, so the user sees the
1205
+ * loop escalating before the watchdog pauses it. Memory-only latch (the
1206
+ * restored state re-notifies once, which is the useful behavior). */
1207
+ function notifyBlockedEscalation(ctx: ExtensionContext, ex: ExecState): void {
1208
+ const level = ex.blocked?.escalatedRounds ?? 0;
1209
+ if (!ex.blocked || level === 0 || ex.blockedNotifiedLevel === level) return;
1210
+ ex.blockedNotifiedLevel = level;
1211
+ const ids = blockedTaskIds(ex);
1212
+ try {
1213
+ messaging().sendMessage(
1214
+ {
1215
+ customType: EXECUTION_BLOCKED_CUSTOM_TYPE,
1216
+ content: `**pi-plans: execution blocked (wake ${level}/${STALL_MAX_ROUNDS})** — review round ${ex.blocked.round + 1} cannot start: ${ids.join(", ")} ${ids.length === 1 ? "was" : "were"} reopened by round ${ex.blocked.round} and ${ids.length === 1 ? "is" : "are"} still open. Close them with \`plans_update_task\` (the review starts by itself once every task is terminal).`,
1217
+ display: true,
1218
+ },
1219
+ { triggerTurn: false },
1220
+ );
1221
+ } catch {
1222
+ /* best-effort: the wake below still carries the same blocker */
1223
+ }
1224
+ }
1225
+
1226
+ /** v0.9.2: the pause reason names the real blocker instead of the watchdog's
1227
+ * own metric. Single line, never a review-cap prefix (`isReviewCapPause`
1228
+ * matches "execution review exhausted" / "completion audit exhausted"). */
1229
+ function blockedPauseReason(ex: ExecState): string {
1230
+ const ids = blockedTaskIds(ex);
1231
+ const round = ex.blocked?.round ?? ex.audit.rounds;
1232
+ const head = ids.slice(0, 3).join(", ");
1233
+ const more = ids.length > 3 ? `, +${ids.length - 3} more` : "";
1234
+ return `blocked: review round ${round + 1} cannot start — ${ids.length} task(s) reopened by round ${round} still open (${head}${more})`;
899
1235
  }
900
1236
 
901
1237
  function runDirOf(ctx: ExtensionContext): string | null {
@@ -996,11 +1332,21 @@ function freshReviewLane(attempt: number): RefineLaneState {
996
1332
 
997
1333
  /** Fresh controller per round (the controller is one-shot: closed latch,
998
1334
  * overlayPromise bail, terminal-lane early return — reuse drops progress).
999
- * A UI failure must NEVER kill the round itself — best-effort only. */
1335
+ * A UI failure must NEVER kill the round itself — best-effort only. The
1336
+ * overlay forwards unhandled keys so Ctrl+Shift+T keeps working while the
1337
+ * review overlay holds focus (pi-tui has no key bubbling). */
1000
1338
  function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
1001
1339
  if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
1002
1340
  try {
1003
- const controller = new RefineOverlayController("auditor", [{ id: lane.id, label: lane.label }], () => {}, lang ?? "en");
1341
+ const controller = new RefineOverlayController(
1342
+ "auditor",
1343
+ [{ id: lane.id, label: lane.label }],
1344
+ () => {},
1345
+ lang ?? "en",
1346
+ (data) => {
1347
+ if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
1348
+ },
1349
+ );
1004
1350
  controller.seedLane(lane);
1005
1351
  controller.open(refineOverlayContext(ctx), modelLabel);
1006
1352
  return controller;
@@ -1013,16 +1359,27 @@ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "
1013
1359
  * state; inert when no round is in flight. */
1014
1360
  export function reopenReviewOverlay(ctx: ExtensionContext): void {
1015
1361
  if (!execution?.review.inFlight || !reviewLane) return;
1362
+ // Never stack a second overlay on a live one (Ctrl+Shift+R while open).
1363
+ if (reviewOverlay && !reviewOverlay.isClosed()) return;
1016
1364
  const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
1017
1365
  if (controller) reviewOverlay = controller;
1018
1366
  }
1019
1367
 
1020
1368
  function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1369
+ const highs = unresolvedHighFindings(ex);
1021
1370
  const detail =
1022
1371
  ex.audit.failed.length > 0
1023
- ? `failed: ${ex.audit.failed.join(", ")}`
1024
- : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1025
- const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1372
+ ? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
1373
+ : highs.length > 0
1374
+ ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1375
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1376
+ // v0.9.3: the reason names the budget actually in force — the numeric round
1377
+ // count, or the unlimited hard cap that stopped the loop.
1378
+ const scope =
1379
+ activeBudget(ex) === "unlimited"
1380
+ ? `${unlimitedHardCapCeiling(budgetCounters(ex))} rounds (unlimited budget safety cap)`
1381
+ : `${activeBudget(ex)} round${activeBudget(ex) === 1 ? "" : "s"}`;
1382
+ const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${scope} (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the ${formatReviewBudget(activeBudget(ex))} picker; ordinary messages and session restores do not). (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1026
1383
  pauseForStall(ctx, reason);
1027
1384
  // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
1028
1385
  // pauseForStall's ui.notify is optional and absent headless, so the pause
@@ -1033,6 +1390,41 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1033
1390
  );
1034
1391
  }
1035
1392
 
1393
+ /** v0.9.3: the unlimited budget's no-progress valve — three consecutive
1394
+ * committed rounds with an identical outcome signature cannot converge, so
1395
+ * the run pauses fail-closed instead of burning rounds silently. */
1396
+ function pauseReviewNoProgress(ctx: ExtensionContext, ex: ExecState): void {
1397
+ const streak = ex.reviewNoProgress?.streak ?? NO_PROGRESS_MAX_STREAK;
1398
+ const highs = unresolvedHighFindings(ex);
1399
+ const detail =
1400
+ ex.audit.failed.length > 0
1401
+ ? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
1402
+ : highs.length > 0
1403
+ ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1404
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1405
+ const reason = `${REVIEW_NO_PROGRESS_PAUSE_PREFIX} — ${streak} consecutive rounds reported the same outcome (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the budget picker; ordinary messages and session restores do not).`;
1406
+ pauseForStall(ctx, reason);
1407
+ messaging().sendMessage(
1408
+ { customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
1409
+ { triggerTurn: false },
1410
+ );
1411
+ }
1412
+
1413
+ /** v0.9.3: the single gate every round spawn passes through. Returns false
1414
+ * when the run was paused (no round may start). The unlimited valve order is
1415
+ * deliberate: no-progress is checked first so its reason wins when both hold. */
1416
+ function reviewBudgetGate(ctx: ExtensionContext, ex: ExecState): boolean {
1417
+ if (activeBudget(ex) === "unlimited" && noProgressTripped(ex.reviewNoProgress)) {
1418
+ pauseReviewNoProgress(ctx, ex);
1419
+ return false;
1420
+ }
1421
+ if (budgetSpent(ex)) {
1422
+ pauseReviewCap(ctx, ex);
1423
+ return false;
1424
+ }
1425
+ return true;
1426
+ }
1427
+
1036
1428
  async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1037
1429
  if (!execution) return;
1038
1430
  const ex = execution;
@@ -1044,15 +1436,34 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1044
1436
  // Only auditable checks (with task coverage) gate completion; checks that
1045
1437
  // cover no task can never be verified and never block or complete.
1046
1438
  const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
1047
- if (pendingChecks.length === 0) {
1439
+ // v0.9: unresolved high findings block the fast completion path — they
1440
+ // keep the review owed instead (liveness: a high can never be completed
1441
+ // around, only fixed or paused at the cap).
1442
+ if (pendingChecks.length === 0 && unresolvedHighFindings(ex).length === 0) {
1048
1443
  await completeExecution(ctx);
1049
1444
  return;
1050
1445
  }
1051
- if (ex.stall.paused || ex.review.inFlight) return;
1052
- if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1053
- pauseReviewCap(ctx, ex);
1054
- return;
1446
+ if (ex.stall.paused || ex.review.inFlight || ex.review.budgetAsking) return;
1447
+ // v0.9.3: the budget is resolved right here — every task is terminal, the
1448
+ // review is genuinely owed, and this is the last moment before round 1.
1449
+ // No panel → the default applies synchronously (no extra tick, so a
1450
+ // detached settle still shows the round in flight immediately); a panel →
1451
+ // one await, guarded against a second settle by `budgetAsking`.
1452
+ if (ex.reviewBudget === undefined && !reviewBudgetPanelUsable(ctx)) applyDefaultReviewBudget(ctx, ex);
1453
+ if (ex.reviewBudget === undefined) {
1454
+ await askReviewBudgetForRun(ctx, ex);
1455
+ if (execution !== ex || ex.reviewBudget === undefined) return;
1456
+ }
1457
+ if (pendingChecks.length === 0) {
1458
+ // Every verification check is satisfied; only unresolved highs keep the
1459
+ // review owed. With the budget spent there is no round left to buy, so
1460
+ // the run completes and DISCLOSES the highs (v0.9.3, Q-2).
1461
+ if (budgetSpent(ex)) {
1462
+ await completeExecution(ctx);
1463
+ return;
1464
+ }
1055
1465
  }
1466
+ if (!reviewBudgetGate(ctx, ex)) return;
1056
1467
  // Phase transition: the executor is done with its tasks; the review loop
1057
1468
  // owns the run until it converges (or pauses at the cap).
1058
1469
  setRunStatusForReview(ctx, "verifying");
@@ -1075,12 +1486,13 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1075
1486
  let result: AuditRoundResult;
1076
1487
  try {
1077
1488
  result = auditRunnerForTests
1078
- ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
1489
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt, priorFindings: ex.audit.findings })
1079
1490
  : await runCompletionAudit(ctx, {
1080
1491
  planPath: ex.planPath,
1081
1492
  checklist: ex.items,
1082
1493
  tasks: ex.tasks,
1083
1494
  round: round.attempt,
1495
+ priorFindings: ex.audit.findings,
1084
1496
  model: spawn.model,
1085
1497
  thinkingLevel: spawn.thinkingLevel,
1086
1498
  timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
@@ -1160,6 +1572,9 @@ async function handleReviewOutcome(
1160
1572
  passed: [],
1161
1573
  failed: [],
1162
1574
  undeterminable: pendingIds,
1575
+ // v0.9.1 (F-014): the discard report must not understate a
1576
+ // round whose embedded report carries highs.
1577
+ findings: owner.audit.findings,
1163
1578
  discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
1164
1579
  fingerprintCaptured: round.fingerprint,
1165
1580
  fingerprintFound: fingerprintNow,
@@ -1187,113 +1602,265 @@ async function handleReviewOutcome(
1187
1602
  await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
1188
1603
  }
1189
1604
 
1605
+ /** v0.9 (findings-driven fix loop): append plan tasks for high findings no
1606
+ * existing task owns. The reviewer stays read-only — it proposes the title
1607
+ * (proposed-task); this machinery applies it with provenance, so every high
1608
+ * finding always has an owner the executor can close, and the stable F-###
1609
+ * id rides in the task title for traceability. Best-effort: an unwritable
1610
+ * plan must not crash the loop (the finding then stays stranded and the cap
1611
+ * pause surfaces it). Returns the appended task ids. */
1612
+ function appendFindingTasks(ex: ExecState, highs: ReviewFinding[]): string[] {
1613
+ const appended: string[] = [];
1614
+ let next = flattenTaskViews(ex.tasks).reduce((max, t) => {
1615
+ const m = /^Task-(\d+)$/.exec(t.id);
1616
+ return m ? Math.max(max, Number(m[1])) : max;
1617
+ }, 0);
1618
+ const wave = maxWave(ex.tasks) + 1;
1619
+ try {
1620
+ const planText = fs.readFileSync(ex.planPath, "utf8");
1621
+ const lines = planText.split("\n");
1622
+ // Insert inside the Tasks section: before the Execution Waves
1623
+ // subsection when present, else before the FIRST checklist header the
1624
+ // plan uses (v0.9.1 F-013: legacy plans say `## Verifier Checklist`,
1625
+ // and an EOF fallback would land outside every parsed section), else
1626
+ // at EOF; walk back over blank separators so the bullet lands
1627
+ // adjacent to its siblings.
1628
+ let insertAt = lines.length;
1629
+ const wavesIdx = lines.findIndex((l) => /^###\s+Execution Waves/.test(l));
1630
+ const checklistIdx = lines.findIndex((l) => CHECKLIST_HEADERS.some((h) => new RegExp(`^##\\s+${h}`).test(l)));
1631
+ if (wavesIdx !== -1) insertAt = wavesIdx;
1632
+ else if (checklistIdx !== -1) insertAt = checklistIdx;
1633
+ while (insertAt > 0 && lines[insertAt - 1].trim() === "") insertAt--;
1634
+ // v0.9.1 (F-012): reviewer text becomes task-title metadata at parse
1635
+ // time — strip the microsyntax metacharacters (em/en dashes, `--`
1636
+ // separators, the `;` field delimiter) so an embedded token can never
1637
+ // split title from tail or forge fields.
1638
+ const sanitize = (text: string): string => text.replace(/[—–]/g, "-").replace(/-{2,}/g, "-").replace(/;/g, ",");
1639
+ const entries: Array<{ id: string; title: string }> = [];
1640
+ const newLines: string[] = [];
1641
+ for (const h of highs) {
1642
+ next += 1;
1643
+ const id = `Task-${next}`;
1644
+ const title = `fix ${h.id}: ${sanitize(h.proposedTask ?? h.note ?? "address the finding")} (appended by execution review round ${ex.audit.rounds})`;
1645
+ // v0.9.1 (F-006): carry the wave in the bullet tail so a re-parse
1646
+ // restores the same wave the live tree assigned — without it the
1647
+ // appended remediation task fell back to wave 1 on restore and
1648
+ // hijacked the ▸ anchor.
1649
+ newLines.push(`- \`${id}\`: ${title} — wave: ${wave}`);
1650
+ entries.push({ id, title });
1651
+ }
1652
+ lines.splice(insertAt, 0, ...newLines);
1653
+ fs.writeFileSync(ex.planPath, lines.join("\n"), "utf8");
1654
+ // v0.9.1 (F-008): only a successful plan write mints the live tasks —
1655
+ // pushing before the write left checkpoint entries the plan file does
1656
+ // not contain whenever the write failed.
1657
+ for (const entry of entries) {
1658
+ ex.tasks.push({ id: entry.id, title: entry.title, wave, deps: [], files: [], status: "pending", children: [] });
1659
+ appended.push(entry.id);
1660
+ }
1661
+ } catch {
1662
+ /* best-effort: stranded highs surface via the cap pause */
1663
+ }
1664
+ return appended;
1665
+ }
1666
+
1190
1667
  async function commitReviewOutcome(
1191
1668
  ctx: ExtensionContext,
1192
1669
  ex: ExecState,
1193
1670
  round: InFlightReview,
1194
- outcome: { passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null,
1671
+ outcome: AuditOutcome | null,
1195
1672
  pendingIds: string[],
1196
1673
  ): Promise<void> {
1197
1674
  // The budget is charged only when an outcome commits — never on discard
1198
1675
  // or cancellation (CF2-003).
1199
1676
  ex.audit.rounds = round.budgetRound;
1677
+ // v0.9.3: the run-cumulative counter never resets — it is what bounds an
1678
+ // unlimited budget across grants (Q-4).
1679
+ ex.reviewRoundsTotal += 1;
1200
1680
  const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
1201
1681
  const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
1202
1682
  // Anything the round neither passed nor failed is undeterminable: the
1203
1683
  // report omitted the check, spelled the verdict unreadably, or the
1204
1684
  // subagent never ran.
1205
1685
  const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
1686
+ // v0.9: the newest round's reported findings ARE the unresolved set
1687
+ // (stable ids — a problem is resolved only by no longer being reported).
1688
+ // v0.9.1 (F-001): only a REAL parsed report is authoritative. A round that
1689
+ // produced no report at all — spawn failure (outcome === null) or the
1690
+ // two-consecutive-discard synthesis — must PRESERVE the unresolved set:
1691
+ // clearing it let the vacuous completion guard (empty pendingIds) mark a
1692
+ // run done with its high finding silently dropped.
1693
+ const reported = outcome !== null && outcome.findings !== undefined;
1694
+ const findings = reported ? outcome.findings : ex.audit.findings;
1695
+ ex.audit.findings = findings;
1696
+ const highs = findings.filter((f) => f.severity === "high");
1697
+ // v0.9.3: the no-progress valve's signature is computed HERE, from the
1698
+ // post-classification triple — a spawn-failure round (`outcome === null`)
1699
+ // and a discard synthesis have no findings array of their own, but the
1700
+ // preserved unresolved set plus the derived verdicts still describe a
1701
+ // concrete, comparable outcome (round-1 F-002).
1702
+ ex.reviewNoProgress = bumpNoProgress(ex.reviewNoProgress, noProgressSignature(failed, undeterminable, highs.map((f) => f.id)));
1703
+ // Findings are actionable only when this round actually reported them;
1704
+ // see the fix-loop branch below (v0.9.1, F-001).
1705
+ const actionableHighs = reported ? highs : [];
1206
1706
  const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
1207
1707
  const reportText = outcome?.report ?? "(review subagent failed to run)";
1708
+ let reportPath: string | null = null;
1208
1709
  {
1209
1710
  const runDir = runDirOf(ctx);
1210
1711
  if (runDir) {
1211
- writeReviewRoundReport(runDir, {
1712
+ reportPath = writeReviewRoundReport(runDir, {
1212
1713
  budgetRound: round.budgetRound,
1213
1714
  attempt: round.attempt,
1214
1715
  outcome: outcome === null
1215
1716
  ? "spawn-failed"
1216
- : failed.length > 0
1717
+ : failed.length > 0 || actionableHighs.length > 0
1217
1718
  ? "failed"
1218
1719
  : undeterminable.length > 0 ? "undeterminable" : "passed",
1219
1720
  passed,
1220
1721
  failed,
1221
1722
  undeterminable,
1723
+ findings,
1222
1724
  fingerprintCaptured: round.fingerprint,
1223
1725
  coveredTaskIds,
1224
1726
  report: reportText,
1225
1727
  });
1226
1728
  }
1227
1729
  }
1730
+ // Partial progress counts: a check affirmed this round is done even when
1731
+ // a sibling failed, so a later round only re-judges what is still open.
1732
+ for (const id of passed) {
1733
+ const item = ex.items.find((candidate) => candidate.id === id);
1734
+ if (item) item.done = true;
1735
+ }
1228
1736
 
1229
- // Fail-closed completion: every pending check affirmatively passed. An
1230
- // all-undeterminable round yields failed === [] — completing here would be
1231
- // fail-open, marking a run done with nothing verified.
1232
- if (passed.length === pendingIds.length) {
1737
+ // v0.9.3 (Q-2/F-003): the exhausted-budget completion. EVERY check this
1738
+ // round still owed affirmed, only unresolved highs remain, and the budget
1739
+ // cannot buy another round — the run completes, tolerating the highs (they
1740
+ // stay in the round report and the checkpoint; `completeExecution`
1741
+ // discloses them). Evaluated BEFORE the fix branch on purpose: the
1742
+ // tolerated round must not roll back tasks, append plan tasks, rewrite the
1743
+ // approved plan, or wake the executor for a run that is about to be done.
1744
+ if (budgetSpent(ex) && passed.length === pendingIds.length && highs.length > 0) {
1233
1745
  ex.audit.failed = [];
1234
1746
  ex.audit.undeterminable = [];
1235
- for (const id of passed) {
1236
- const item = ex.items.find((candidate) => candidate.id === id);
1237
- if (item) item.done = true;
1238
- }
1747
+ ex.blocked = null;
1239
1748
  withExecutionCheckpoint(ctx, (cp) =>
1240
1749
  applyExecutionProgress(cp, {
1241
1750
  tasks: taskProgressMap(ex.tasks),
1242
1751
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1243
- audit: { rounds: ex.audit.rounds, passed: true },
1752
+ blocked: null,
1753
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1754
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1755
+ audit: { rounds: ex.audit.rounds, passed: true, findings },
1244
1756
  }),
1245
1757
  );
1758
+ persist(ctx);
1759
+ messaging().sendMessage(
1760
+ {
1761
+ customType: "pi-plans-review-tolerated",
1762
+ content: `**pi-plans: review budget exhausted (${budgetLabel(ex)}) — completing with ${highs.length} unresolved high finding(s): ${highs.map((f) => f.id).join(", ")}** — every verification check passed; the finding(s) remain in the round reports.`,
1763
+ display: true,
1764
+ },
1765
+ { triggerTurn: false },
1766
+ );
1246
1767
  await completeExecution(ctx);
1247
1768
  return;
1248
1769
  }
1249
- // Partial progress counts: a check affirmed this round is done even when a
1250
- // sibling failed, so a later round only re-judges what is still open.
1251
- for (const id of passed) {
1252
- const item = ex.items.find((candidate) => candidate.id === id);
1253
- if (item) item.done = true;
1254
- }
1255
- ex.audit.failed = failed;
1256
- ex.audit.undeterminable = undeterminable;
1257
- const rolledBack: string[] = [];
1258
- for (const id of failed) {
1259
- rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1260
- }
1261
- // A rollback reopens work other checks were verifying; those checks must
1262
- // stop claiming the run is satisfied there.
1263
- if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1264
- withExecutionCheckpoint(ctx, (cp) =>
1265
- applyExecutionProgress(cp, {
1266
- tasks: taskProgressMap(ex.tasks),
1267
- doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1268
- audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
1269
- }),
1270
- );
1271
- if (rolledBack.length > 0) {
1272
- // Rolling back is itself forward progress for the watchdog, but NOT for
1273
- // audit.rounds: that counter stays monotonic so the repair loop is
1274
- // bounded. The repair belongs to the executor — back to executing.
1275
- ex.stall.rounds = 0;
1276
- ex.stall.lastSnapshot = stallSnapshot();
1277
- setRunStatusForReview(ctx, "executing");
1278
- }
1279
- persist(ctx);
1280
- updateStatusWidget(ctx);
1281
- if (failed.length > 0) {
1282
- // v0.8 wake: per-round one-shot token — exactly one triggerTurn per
1283
- // committed failed outcome, even when the round resolves long after the
1284
- // settle that spawned it. The continuation runtime is deliberately
1285
- // untouched: detached rounds outlive their settle.
1770
+
1771
+ // v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
1772
+ // high finding can never complete the run (liveness). Failed checks and
1773
+ // high findings drive ONE union rollback and exactly one executor wake;
1774
+ // high wins over the undeterminable self-schedule (a finding is
1775
+ // actionable independent of verdict evidence). v0.9.1 (F-001): verdicts
1776
+ // are authoritative whenever a report exists, but the FINDINGS-driven
1777
+ // half of the branch needs a findings-bearing report — a no-findings
1778
+ // round (spawn failure, discard synthesis, legacy shape) preserves the
1779
+ // unresolved set and self-schedules instead of rolling back on it.
1780
+ if (failed.length > 0 || actionableHighs.length > 0) {
1781
+ ex.audit.failed = failed;
1782
+ ex.audit.undeterminable = undeterminable;
1783
+ const rolledBack: string[] = [];
1784
+ for (const id of failed) {
1785
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1786
+ }
1787
+ // Invalidation is the VC-fail invariant ONLY: a check re-verifies its
1788
+ // reopened tasks when a FAILED check rolled them back. A pure
1789
+ // finding-driven rollback keeps earlier passes — the findings channel
1790
+ // itself re-examines the repaired work next round (stable ids).
1791
+ if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1792
+ const knownIds = new Set(coveredTaskIds);
1793
+ const mappedHighIds = [...new Set(actionableHighs.flatMap((h) => h.taskIds).filter((id) => knownIds.has(id)))];
1794
+ const highRolledBack = findingsRollbackSet(ex.tasks, mappedHighIds);
1795
+ const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
1796
+ const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
1797
+ const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
1798
+ // v0.9.2: capture the round's rollback set HERE. The reopen helpers above
1799
+ // mutate the tree and report only flipped nodes, so this is the only
1800
+ // moment the authoritative provenance exists; the still-open subset is
1801
+ // what blocks the next round and what wakes/pauses name explicitly.
1802
+ ex.blocked = allRolledBack.length > 0
1803
+ ? {
1804
+ rolledBack: [...allRolledBack],
1805
+ tasks: blockedReviewTasks(ex.tasks, allRolledBack),
1806
+ round: ex.audit.rounds,
1807
+ escalatedRounds: 0,
1808
+ since: utcNow(),
1809
+ }
1810
+ : null;
1811
+ ex.blockedWakeTasks = undefined;
1812
+ withExecutionCheckpoint(ctx, (cp) => {
1813
+ // v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
1814
+ // re-stamp the checkpoint's plan identity in the same revision so a
1815
+ // later /resume-plans accepts the amended plan instead of rejecting
1816
+ // it as plan-mismatch (which would cost a full re-approval).
1817
+ const amendedCp = amended.length > 0
1818
+ ? applyExecutionPlanAmended(cp, planIdentityOf(ex.planPath, cp.plan?.version ?? 1), ex.audit.rounds)
1819
+ : cp;
1820
+ return applyExecutionProgress(amendedCp, {
1821
+ tasks: taskProgressMap(ex.tasks),
1822
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1823
+ blocked: ex.blocked,
1824
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1825
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1826
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
1827
+ });
1828
+ });
1829
+ if (allRolledBack.length > 0 || amended.length > 0) {
1830
+ // Rolling back (or appending) is itself forward progress for the
1831
+ // watchdog, but NOT for audit.rounds: that counter stays monotonic
1832
+ // so the repair loop is bounded. The repair belongs to the
1833
+ // executor — back to executing.
1834
+ ex.stall.rounds = 0;
1835
+ ex.stall.lastSnapshot = stallSnapshot();
1836
+ setRunStatusForReview(ctx, "executing");
1837
+ }
1838
+ persist(ctx);
1839
+ updateStatusWidget(ctx);
1840
+ // v0.8 wake, generalized (v0.9): per-round one-shot token — exactly one
1841
+ // triggerTurn per committed fix-needing outcome, even when the round
1842
+ // resolves long after the settle that spawned it.
1286
1843
  if (!round.wakeSent) {
1287
1844
  round.wakeSent = true;
1288
1845
  const openTasks = flattenTaskViews(ex.tasks)
1289
1846
  .filter((task) => !taskIsTerminal(task))
1290
1847
  .map((task) => task.id);
1291
- const stranded = rolledBack.length === 0;
1292
- const content = `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the review automatically.` : ""}`;
1848
+ const stranded = allRolledBack.length === 0 && amended.length === 0;
1849
+ const highLines = actionableHighs.map((h) => `- ${h.id}${h.taskIds.length ? ` (${h.taskIds.join(", ")})` : ""}: ${h.note}`).join("\n");
1850
+ const reportRef = reportPath
1851
+ ? `Full round report: ${reportPath}`
1852
+ : `Full round report (run dir unwritable — inline):\n\n---\n${reportText.slice(0, 4000)}`;
1853
+ // v0.9.1 (F-004): a pure VC-fail round keeps the v0.8 lead — never
1854
+ // announce "0 high-severity findings" over an empty block.
1855
+ const findingsLead = actionableHighs.length > 0
1856
+ ? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
1857
+ : `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
1858
+ const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
1859
+ const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${budgetSpent(ex) ? (activeBudget(ex) === "unlimited" ? ` This was round ${ex.reviewRoundsTotal} against the unlimited budget's ${unlimitedHardCapCeiling(budgetCounters(ex))}-round safety cap: the next terminal-task cycle pauses the run for review.` : ` This was round ${ex.audit.rounds} of ${budgetLabel(ex)}: the next terminal-task cycle pauses the run for review (or completes if every check passed).`) : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
1293
1860
  messaging().sendMessage(
1294
1861
  {
1295
1862
  customType: "pi-plans-audit-failed",
1296
- content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}\n\n---\n${reportText.slice(0, 4000)}`,
1863
+ content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}`,
1297
1864
  display: true,
1298
1865
  },
1299
1866
  { triggerTurn: true },
@@ -1301,8 +1868,48 @@ async function commitReviewOutcome(
1301
1868
  }
1302
1869
  return; // The agent repairs; the next settle re-enters the loop.
1303
1870
  }
1871
+
1872
+ // Fail-closed completion: every pending check affirmatively passed AND no
1873
+ // high finding remains. The highs guard is explicit (v0.9.1, F-001): a
1874
+ // no-report round skips the fix branch above, so this is the last line
1875
+ // against vacuously completing an empty-pendingIds round with an
1876
+ // unresolved high. An all-undeterminable round yields failed === [] —
1877
+ // completing here would be fail-open, marking a run done with nothing
1878
+ // verified.
1879
+ if (passed.length === pendingIds.length && highs.length === 0) {
1880
+ ex.audit.failed = [];
1881
+ ex.audit.undeterminable = [];
1882
+ // v0.9.2: a passing audit ends the blocker — the review consumed it.
1883
+ ex.blocked = null;
1884
+ withExecutionCheckpoint(ctx, (cp) =>
1885
+ applyExecutionProgress(cp, {
1886
+ tasks: taskProgressMap(ex.tasks),
1887
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1888
+ blocked: null,
1889
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1890
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1891
+ audit: { rounds: ex.audit.rounds, passed: true, findings },
1892
+ }),
1893
+ );
1894
+ await completeExecution(ctx);
1895
+ return;
1896
+ }
1897
+
1304
1898
  // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1305
1899
  // wake, no message; the dashboard/overlay carries the round counter.
1900
+ ex.audit.failed = failed;
1901
+ ex.audit.undeterminable = undeterminable;
1902
+ withExecutionCheckpoint(ctx, (cp) =>
1903
+ applyExecutionProgress(cp, {
1904
+ tasks: taskProgressMap(ex.tasks),
1905
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1906
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1907
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1908
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
1909
+ }),
1910
+ );
1911
+ persist(ctx);
1912
+ updateStatusWidget(ctx);
1306
1913
  await maybeContinueReview(ctx, ex);
1307
1914
  }
1308
1915
 
@@ -1310,10 +1917,11 @@ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promis
1310
1917
  if (execution !== ex) return;
1311
1918
  if (!reviewOwed(ex)) return;
1312
1919
  if (ex.stall.paused) return;
1313
- if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1314
- pauseReviewCap(ctx, ex);
1315
- return;
1316
- }
1920
+ // The budget gate owns every pause (numeric exhaustion, the unlimited hard
1921
+ // cap, and the unlimited no-progress valve). An exhausted budget with every
1922
+ // check satisfied has already completed inside commitReviewOutcome; here
1923
+ // the review is still owed, so a spent budget pauses.
1924
+ if (!reviewBudgetGate(ctx, ex)) return;
1317
1925
  await startReviewRound(ctx);
1318
1926
  }
1319
1927
 
@@ -1899,7 +2507,10 @@ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1899
2507
  // This also bounds the zero-input continue loop in agent_before_settle.
1900
2508
  if (ex.stall.paused) return false;
1901
2509
  if (ex.review.inFlight) return false;
1902
- return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
2510
+ // The budget panel is open: the review is being resolved, not owed anew.
2511
+ if (ex.review.budgetAsking) return false;
2512
+ return allTasksTerminal(ex.tasks)
2513
+ && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
1903
2514
  }
1904
2515
 
1905
2516
  /** v0.7.1: record that this settle already ran (or declined) its audit, so a
@@ -1984,6 +2595,33 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
1984
2595
  if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
1985
2596
  runtime.handled = true;
1986
2597
  const ex = runtime.owner;
2598
+ // v0.9.2: while a failed round's tasks are still open, the next round
2599
+ // cannot start — and this ladder must NOT ride `stall.rounds`, which any
2600
+ // successful tool call resets: an executor that investigates but never
2601
+ // closes the reopened tasks has to escalate (and eventually pause) anyway.
2602
+ const blockedIds = blockedTaskIds(ex);
2603
+ if (ex.blocked && blockedIds.length > 0) {
2604
+ // Progress = fewer blockers than at the PREVIOUS blocked wake. The
2605
+ // stored `blocked.tasks` cannot serve as that baseline: every task
2606
+ // close re-syncs it, so it always equals the live set and the ladder
2607
+ // would only ever increment (review round 1, F-001). A same-size but
2608
+ // different set counts as no progress; a new member counts as none.
2609
+ const baseline = ex.blockedWakeTasks;
2610
+ const progressed = baseline !== undefined && blockedIds.length < baseline.length;
2611
+ ex.blockedWakeTasks = [...blockedIds];
2612
+ ex.blocked.escalatedRounds = progressed ? 0 : ex.blocked.escalatedRounds + 1;
2613
+ ex.blocked.tasks = blockedIds;
2614
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { blocked: ex.blocked }));
2615
+ if (ex.blocked.escalatedRounds >= STALL_MAX_ROUNDS) {
2616
+ pauseForStall(ctx, blockedPauseReason(ex));
2617
+ return;
2618
+ }
2619
+ notifyBlockedEscalation(ctx, ex);
2620
+ persist(ctx);
2621
+ updateStatusWidget(ctx);
2622
+ sendContinuationWake(ctx, runtime);
2623
+ return;
2624
+ }
1987
2625
  const snapshot = stallSnapshot();
1988
2626
  const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
1989
2627
  ex.stall.lastSnapshot = snapshot;
@@ -2003,8 +2641,10 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
2003
2641
 
2004
2642
  export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
2005
2643
  // v0.6.1: continuation wakes are one-shot; stale ones (including the
2006
- // legacy v0.6.0 goal-wait type) never replay after a restart.
2644
+ // legacy v0.6.0 goal-wait type) never replay after a restart. v0.9.2: the
2645
+ // visible escalated-blocked line is one-shot for the same reason.
2007
2646
  return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
2647
+ && message.customType !== EXECUTION_BLOCKED_CUSTOM_TYPE
2008
2648
  && message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
2009
2649
  }
2010
2650
 
@@ -2013,18 +2653,20 @@ export function filterContinuationMessages<T extends { customType?: string; deta
2013
2653
  }
2014
2654
 
2015
2655
  /** Called for genuine user input or an explicit same-execution resume.
2016
- * v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
2017
- * refill the five-round budget (CF2-004); only /plans-execute
2656
+ * v0.8: a REVIEW pause is never lifted here — ordinary input must not
2657
+ * refill the review budget (CF2-004); only /plans-execute
2018
2658
  * (resumeActiveExecution) is the explicit confirmation surface. Genuine
2019
2659
  * stall pauses still clear on input as before. */
2020
2660
  export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
2021
2661
  const ex = getExecution();
2022
2662
  if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
2023
- if (isReviewCapPause(ex.stall.pausedReason)) {
2663
+ if (isReviewPauseReason(ex.stall.pausedReason)) {
2024
2664
  // Surfaced once per input so the user is not left guessing why the run
2025
2665
  // stays paused; the pause itself and the budget survive untouched.
2666
+ // v0.9.3: the note names the budget actually in force and the fact that
2667
+ // the confirmation re-opens the picker.
2026
2668
  ctx.ui.notify?.(
2027
- "pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
2669
+ `pi-plans: the review budget is exhausted (${budgetLabel(ex)}) — run /plans-execute to grant a fresh budget (it re-opens the round-count picker; that confirmation is the only surface that does).`,
2028
2670
  "warning",
2029
2671
  );
2030
2672
  return false;
@@ -2033,18 +2675,61 @@ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
2033
2675
  ex.stall.pausedReason = undefined;
2034
2676
  ex.stall.rounds = 0;
2035
2677
  ex.stall.lastSnapshot = stallSnapshot();
2036
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
2678
+ // v0.9.2: a resume grants a fresh escalation ladder (the blocker itself is
2679
+ // still recorded, so the next wake names it) and never a fresh review
2680
+ // budget — that stays `/plans-execute`-only.
2681
+ if (ex.blocked) {
2682
+ ex.blocked.escalatedRounds = 0;
2683
+ ex.blockedWakeTasks = undefined;
2684
+ }
2685
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null, blocked: ex.blocked }));
2037
2686
  persist(ctx);
2038
2687
  updateStatusWidget(ctx);
2039
2688
  return true;
2040
2689
  }
2041
2690
 
2042
- export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2043
- // v0.8: /plans-execute is THE explicit confirmation surface for a
2044
- // review-cap pause — the only place a fresh five-round budget is granted
2045
- // (Q-confirm-surface). Ordinary input and session restores never refill.
2691
+ /** Outcome of an explicit `/plans-execute` resume, so the calling tool can
2692
+ * report what actually happened (v0.9.3: the grant may re-open the picker). */
2693
+ export interface ResumeOutcome {
2694
+ resumed: boolean;
2695
+ /** Set when the pause was lifted by a budget grant. */
2696
+ grantedBudget?: ReviewBudget;
2697
+ /** True when the user closed the budget panel — the pause stands. */
2698
+ budgetDeclined?: boolean;
2699
+ }
2700
+
2701
+ export async function resumeActiveExecution(ctx: ExtensionContext): Promise<ResumeOutcome> {
2702
+ // v0.8: /plans-execute is THE explicit confirmation surface for a review
2703
+ // pause — the only place a fresh budget is granted (Q-confirm-surface).
2704
+ // Ordinary input and session restores never refill.
2705
+ // v0.9.3: the grant re-opens the budget picker with the current value
2706
+ // preselected; Esc keeps the run paused (an explicit confirmation is the
2707
+ // only way forward). Headless sessions, which have no panel to show, keep
2708
+ // the current budget instead of stranding the run.
2046
2709
  const pausedEx = getExecution();
2047
- if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
2710
+ if (pausedEx?.stall.paused && isReviewPauseReason(pausedEx.stall.pausedReason)) {
2711
+ const previous = activeBudget(pausedEx);
2712
+ let granted = previous;
2713
+ if (reviewBudgetPanelUsable(ctx)) {
2714
+ const picked = await askReviewBudget(ctx, pausedEx.uiLanguage, previous);
2715
+ if (picked === null) {
2716
+ messaging().sendMessage(
2717
+ {
2718
+ customType: "pi-plans-review-budget-declined",
2719
+ content: `**pi-plans: review budget unchanged (${formatReviewBudget(previous)})** — the run stays paused. Run /plans-execute and pick a round count to continue.`,
2720
+ display: true,
2721
+ },
2722
+ { triggerTurn: false },
2723
+ );
2724
+ return { resumed: false, budgetDeclined: true };
2725
+ }
2726
+ granted = picked;
2727
+ // Q-4: only a grant that lands on `unlimited` lifts the hard cap —
2728
+ // the cumulative counter itself never resets.
2729
+ if (granted === "unlimited") {
2730
+ pausedEx.reviewCapExtension += UNLIMITED_HARD_CAP;
2731
+ }
2732
+ }
2048
2733
  pausedEx.stall.paused = false;
2049
2734
  pausedEx.stall.pausedReason = undefined;
2050
2735
  pausedEx.stall.rounds = 0;
@@ -2052,26 +2737,45 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2052
2737
  pausedEx.audit.rounds = 0;
2053
2738
  pausedEx.audit.failed = [];
2054
2739
  pausedEx.audit.undeterminable = [];
2740
+ // v0.9.3: a fresh budget window also resets the no-progress valve.
2741
+ pausedEx.reviewNoProgress = undefined;
2742
+ // v0.9.2: the explicit confirmation also refreshes the blocked ladder
2743
+ // (the blocker set itself survives — it still names what stays open).
2744
+ if (pausedEx.blocked) {
2745
+ pausedEx.blocked.escalatedRounds = 0;
2746
+ pausedEx.blockedWakeTasks = undefined;
2747
+ }
2748
+ // v0.9: the fresh budget inherits unresolved findings (stable ids keep
2749
+ // counting) — only the round counter resets.
2055
2750
  withExecutionCheckpoint(ctx, (cp) =>
2056
2751
  applyExecutionProgress(cp, {
2057
2752
  tasks: taskProgressMap(pausedEx.tasks),
2058
- audit: { rounds: 0, lastResult: undefined },
2753
+ audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
2754
+ blocked: pausedEx.blocked,
2755
+ reviewBudget: granted,
2756
+ reviewBudgetDefaulted: pausedEx.reviewBudgetDefaulted === true,
2757
+ reviewRoundsTotal: pausedEx.reviewRoundsTotal,
2758
+ reviewCapExtension: pausedEx.reviewCapExtension,
2759
+ reviewNoProgress: null,
2059
2760
  pausedReason: null,
2060
2761
  }),
2061
2762
  );
2062
2763
  persist(ctx);
2063
2764
  updateStatusWidget(ctx);
2765
+ const grantedNote = granted === "unlimited"
2766
+ ? `unlimited review budget granted (hard cap now ${unlimitedHardCapCeiling(budgetCounters(pausedEx))} rounds; the run has spent ${pausedEx.reviewRoundsTotal})`
2767
+ : `fresh ${formatReviewBudget(granted)}-round review budget granted`;
2064
2768
  messaging().sendMessage(
2065
2769
  {
2066
2770
  customType: "pi-plans-review-budget-granted",
2067
- content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
2771
+ content: `**pi-plans: ${grantedNote}** — the execution review resumes now.`,
2068
2772
  display: true,
2069
2773
  },
2070
2774
  { triggerTurn: false },
2071
2775
  );
2072
2776
  const grantChain = launchReviewRound(ctx);
2073
2777
  if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
2074
- return true;
2778
+ return { resumed: true, grantedBudget: granted };
2075
2779
  }
2076
2780
  // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
2077
2781
  // to `return false` here, so `/plans-execute` answered "already executing"
@@ -2082,15 +2786,15 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2082
2786
  latchAuditThisSettle();
2083
2787
  const chain = launchReviewRound(ctx);
2084
2788
  if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
2085
- return true;
2789
+ return { resumed: true };
2086
2790
  }
2087
- if (!resumeGoalWaitIfPaused(ctx)) return false;
2791
+ if (!resumeGoalWaitIfPaused(ctx)) return { resumed: false };
2088
2792
  const runtime = currentContinuationRuntime(ctx)!;
2089
2793
  if (canWakeExecution(ctx, runtime)) {
2090
2794
  runtime.handled = true;
2091
2795
  sendContinuationWake(ctx, runtime);
2092
2796
  }
2093
- return true;
2797
+ return { resumed: true };
2094
2798
  }
2095
2799
 
2096
2800
  export async function completeExecution(ctx: ExtensionContext): Promise<void> {
@@ -2102,6 +2806,19 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2102
2806
  const summary = flat
2103
2807
  .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
2104
2808
  .join("\n");
2809
+ // v0.9: residual (non-high) findings are summarized, never silent — the
2810
+ // loop converged because nothing high-blocking remained.
2811
+ const residualFindings = execution.audit.findings.filter((f) => f.severity !== "high");
2812
+ const residualNote = residualFindings.length > 0
2813
+ ? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
2814
+ : "";
2815
+ // v0.9.3 (Q-2, round-1 F-004): an exhausted budget may complete WITH
2816
+ // unresolved high findings — the completion surface must say so instead of
2817
+ // claiming "execution review passed".
2818
+ const toleratedHighs = execution.audit.findings.filter((f) => f.severity === "high");
2819
+ const toleratedNote = toleratedHighs.length > 0
2820
+ ? `\n\n⚠️ The review budget was exhausted (${budgetLabel(execution)}) before these high-severity finding(s) could be resolved: ${toleratedHighs.map((f) => f.id).join(", ")} — every verification check passed; the finding(s) remain in the round reports under the run directory.`
2821
+ : "";
2105
2822
  const planPath = execution.planPath;
2106
2823
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
2107
2824
  execution = null;
@@ -2111,7 +2828,9 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2111
2828
  messaging().sendMessage(
2112
2829
  {
2113
2830
  customType: "pi-plans-complete",
2114
- content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}`,
2831
+ content: toleratedHighs.length > 0
2832
+ ? `**Plan complete (review budget exhausted).** ⚠️ \`${planPath}\` — all verification checks passed; ${toleratedHighs.length} high-severity finding(s) stayed unresolved.\n\n${summary}${residualNote}${toleratedNote}`
2833
+ : `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
2115
2834
  display: true,
2116
2835
  },
2117
2836
  { triggerTurn: false },
@@ -2146,13 +2865,32 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
2146
2865
  const rollbackNote = execution.audit.failed.length > 0
2147
2866
  ? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
2148
2867
  : "";
2868
+ const unresolvedHighs = unresolvedHighFindings(execution);
2869
+ const highFindingsNote = unresolvedHighs.length > 0
2870
+ ? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
2871
+ : "";
2872
+ // v0.9.2: the wake must answer "is the review running?" explicitly. The
2873
+ // previous text left the executor waiting for a round that cannot start
2874
+ // while the tree is open (the exact stall this feature fixes).
2875
+ const blockedIds = blockedTaskIds(execution);
2876
+ const blockedRound = execution.blocked?.round ?? execution.audit.rounds;
2877
+ const outstanding = reviewOutstanding(execution);
2878
+ const reviewLine = execution.review.inFlight
2879
+ ? `\nReview: round ${execution.audit.rounds + 1}/${budgetLabel(execution)} IS RUNNING (read-only reviewer verifying) — do not wait on it and do not re-close tasks for it.`
2880
+ : blockedIds.length > 0 && outstanding
2881
+ ? `\nReview: NO round is running — the task tree is not terminal, so the review cannot start. It starts by itself the moment every task is terminal (round ${execution.audit.rounds + 1}/${budgetLabel(execution)}).`
2882
+ : "";
2883
+ const escalated = (execution.blocked?.escalatedRounds ?? 0) > 0;
2884
+ const blockedLine = blockedIds.length > 0 && outstanding
2885
+ ? `\nBLOCKED — ${blockedIds.length} task(s) reopened by round ${blockedRound} are still open:\n${blockedIds.map((id) => `- ${id} (reopened by round ${blockedRound}) — close with \`plans_update_task\`: status "complete" with evidence, or "skipped" with a skipReason. Closing a child does NOT close its parent; a parent with an open child is not terminal.`).join("\n")}${escalated ? `\nThis is blocked wake ${execution.blocked?.escalatedRounds}/${STALL_MAX_ROUNDS}: if these tasks stay open, the watchdog pauses the run and a human has to resume it.` : ""}`
2886
+ : "";
2149
2887
  return `[PI-PLANS EXECUTION — write access enabled]
2150
2888
  Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
2151
2889
 
2152
2890
  Current wave ${currentWave} open tasks:
2153
2891
  ${waveList}
2154
2892
 
2155
- Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
2893
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}${reviewLine}${blockedLine}
2156
2894
 
2157
2895
  ${graphLine}
2158
2896
 
@@ -2160,6 +2898,8 @@ Execution rules:
2160
2898
  - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
2161
2899
  - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
2162
2900
  - Close subtasks before their parent; a parent is auditable only when every child is terminal.
2901
+ - After a FAILED review round, every reopened task must be re-closed with fresh evidence — PARENTS INCLUDED. Closing a task's children does NOT close the task: a parent that still has an open child is not terminal, and the review cannot start until the whole tree is terminal.
2902
+ - NEVER wait for the review. While any task is open, the review is not running and nothing will re-close tasks for you: an open task is your work queue — close it with evidence or skip it with a skipReason.
2163
2903
  - When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
2164
2904
  - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
2165
2905
  - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
@@ -2217,7 +2957,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2217
2957
  planTasks = parsePlanTasks(planText);
2218
2958
  const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
2219
2959
  tasks = buildTaskView(planTasks, snapshotProgress);
2220
- items = parseChecklist(planText);
2960
+ // Fresh text wins, but the satisfied state survives the re-parse: a
2961
+ // mid-loop session restore (v0.9.1, found via F-001's self-schedule
2962
+ // path) must not hand the next round a brief that re-judges checks an
2963
+ // earlier round already passed — that burned budget on every /reload.
2964
+ const snapDone = new Set(snapshot.items.filter((c) => c.done).map((c) => c.id));
2965
+ items = parseChecklist(planText).map((item) => (snapDone.has(item.id) ? { ...item, done: true } : item));
2221
2966
  } catch {
2222
2967
  tasks = snapshot.tasks;
2223
2968
  }
@@ -2234,14 +2979,27 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2234
2979
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
2235
2980
  uiLanguage: resolveUiLanguage(ctx.cwd),
2236
2981
  stall: { ...snapshot.stall, lastSnapshot: null },
2982
+ // v0.9.2: a snapshot taken by a pre-feature build carries the live failed
2983
+ // set but no blocker record; reconstruct it so the very first wake after
2984
+ // a /reload (the recovery path for an already-stuck run) names the tasks.
2985
+ blocked: snapshot.blocked ?? blockedFromFailedIds(snapshot.audit?.failed ?? [], snapshot.audit?.rounds ?? 0, tasks, items),
2237
2986
  audit: {
2238
2987
  rounds: snapshot.audit?.rounds ?? 0,
2239
2988
  failed: snapshot.audit?.failed ?? [],
2240
2989
  undeterminable: [],
2990
+ findings: toReviewFindings(snapshot.audit?.findings),
2241
2991
  running: false,
2242
2992
  },
2243
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
2993
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
2244
2994
  auditLatch: { auditedThisSettle: false, activity: 0 },
2995
+ // v0.9.3 (round-1 F-005): the snapshot's budget/counters survive a
2996
+ // /reload — an unset budget stays unset (the picker runs before round 1),
2997
+ // a legacy snapshot resolves to the 5-round bound via the audit counter.
2998
+ reviewBudget: resolveStoredBudget(snapshot.reviewBudget, snapshot.audit?.rounds ?? 0),
2999
+ reviewBudgetDefaulted: snapshot.reviewBudgetDefaulted === true,
3000
+ reviewRoundsTotal: snapshot.reviewRoundsTotal ?? 0,
3001
+ reviewCapExtension: snapshot.reviewCapExtension ?? 0,
3002
+ reviewNoProgress: snapshot.reviewNoProgress ?? undefined,
2245
3003
  };
2246
3004
  execution.stall.lastSnapshot = stallSnapshot();
2247
3005
  resetContinuationRuntime(ctx);