pi-plans 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/exec.ts CHANGED
@@ -66,6 +66,8 @@ import {
66
66
  sha256File,
67
67
  StaleCheckpointError,
68
68
  type ExecutionApproval,
69
+ type ExecutionBlocked,
70
+ type ExecutionCheckpoint,
69
71
  type WorkflowCheckpoint,
70
72
  } from "./workflow-state.ts";
71
73
  import { graphBlockForExecutor } from "./code-graph/prompts.ts";
@@ -82,12 +84,14 @@ import {
82
84
  allTasksTerminal,
83
85
  auditRollbackSet,
84
86
  auditableChecks,
87
+ blockedReviewTasks,
85
88
  buildTaskView,
86
89
  currentTask,
87
90
  findingsRollbackSet,
88
91
  flattenTaskViews,
89
92
  invalidateChecksForRolledBackTasks,
90
93
  maxWave,
94
+ rollbackCoverageIds,
91
95
  taskIsTerminal,
92
96
  taskProgress,
93
97
  taskProgressMap,
@@ -103,7 +107,27 @@ import {
103
107
  renderDashboardLines,
104
108
  renderDashboardTreeLines,
105
109
  } from "./dashboard.ts";
106
- import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
110
+ import { presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
111
+ import {
112
+ DEFAULT_REVIEW_BUDGET,
113
+ LEGACY_REVIEW_MAX_ROUNDS,
114
+ NO_PROGRESS_MAX_STREAK,
115
+ REVIEW_CAP_PAUSE_PREFIX,
116
+ REVIEW_NO_PROGRESS_PAUSE_PREFIX,
117
+ UNLIMITED_HARD_CAP,
118
+ askReviewBudget,
119
+ budgetExhausted,
120
+ bumpNoProgress,
121
+ formatReviewBudget,
122
+ isReviewPauseReason,
123
+ noProgressSignature,
124
+ noProgressTripped,
125
+ resolveStoredBudget,
126
+ reviewBudgetPanelAvailable,
127
+ unlimitedHardCapCeiling,
128
+ type NoProgressState,
129
+ type ReviewBudget,
130
+ } from "./review-budget.ts";
107
131
  import type { ReviewFindingRecord } from "./workflow-state.ts";
108
132
  import { staleReloadHint as probeStaleReload } from "./staleness.ts";
109
133
  import { messaging } from "./messaging.ts";
@@ -137,13 +161,48 @@ export interface ExecState {
137
161
  /** Execution-review loop (v0.8), memory-only: the attempt index names the
138
162
  * per-round report files; consecutiveDiscards bounds the fingerprint
139
163
  * re-run loop; inFlight owns the round's abort lifecycle. */
140
- review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
164
+ review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null; budgetAsking?: boolean };
141
165
  /** Per-settle audit latch (v0.7.1): a settled round fires the completion
142
166
  * audit at most once, so the turn_end / agent_before_settle / resume entry
143
167
  * points cannot double-consume a round when several land in one settle.
144
168
  * Created on demand by auditLatchOf(); every construction path may omit it. */
145
169
  auditLatch?: { auditedThisSettle: boolean; activity: number };
146
- }
170
+ /** v0.9.2: outstanding blocker from the newest failed review round — the
171
+ * authoritative rollback set captured at commit time plus the still-open
172
+ * tasks among it. Memory mirror of `execution.blocked`; null = nothing
173
+ * blocks the review. */
174
+ blocked: ExecBlocked | null;
175
+ /** v0.9.2: blocker ids as observed at the PREVIOUS blocked wake — the
176
+ * ladder's progress baseline. Memory-only: `blocked.tasks` is re-synced by
177
+ * `persistTaskProgress` on every task close (so it stays fresh for the
178
+ * resume brief), which would make a comparison against it meaningless.
179
+ * Absent after a restore/resume → the next blocked wake restarts the count. */
180
+ blockedWakeTasks?: string[];
181
+ /** v0.9.2: highest escalation level already surfaced as a visible system
182
+ * line. Memory-only (never persisted/snapshotted) so a restored session
183
+ * re-notifies once. */
184
+ blockedNotifiedLevel?: number;
185
+ /** v0.9.3: the per-run execution-review budget. Resolved exactly once —
186
+ * right before round 1 — from the picker or the no-UI default, then
187
+ * persisted. `undefined` = not decided yet (never conflated with a legacy
188
+ * checkpoint, which `loadExecutionFromCheckpoint` resolves to 5). */
189
+ reviewBudget?: ReviewBudget;
190
+ /** v0.9.3: the budget came from the fallback (no UI/headless/auto-approve),
191
+ * not from a user pick — the expanded dashboard row marks it `(default)`
192
+ * and the resume brief repeats the note. Persisted so a later session can
193
+ * still tell the difference. */
194
+ reviewBudgetDefaulted?: boolean;
195
+ /** v0.9.3: committed review rounds across the whole run (never reset by a
196
+ * grant) — the unlimited budget's cumulative hard-cap counter. */
197
+ reviewRoundsTotal: number;
198
+ /** v0.9.3: rounds added to the unlimited hard cap by explicit grants. */
199
+ reviewCapExtension: number;
200
+ /** v0.9.3: no-progress valve state (unlimited budget only). */
201
+ reviewNoProgress?: NoProgressState;
202
+ }
203
+
204
+ /** Live blocker record (see `ExecutionBlocked` in workflow-state.ts). */
205
+ type ExecBlocked = ExecutionBlocked;
147
206
 
148
207
  /** One in-flight review round: owns its abort lifecycle, its fingerprint of
149
208
  * the audited subject, and the per-round one-shot wake token (v0.8). */
@@ -173,6 +232,8 @@ const STALL_MAX_ROUNDS = 3;
173
232
  let execution: ExecState | null = null;
174
233
 
175
234
  export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
235
+ /** v0.9.2: visible escalation line — never replays after a restart. */
236
+ export const EXECUTION_BLOCKED_CUSTOM_TYPE = "pi-plans-exec-blocked";
176
237
  /** Legacy v0.6.0 continuation message type — filtered on restore. */
177
238
  const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
178
239
 
@@ -239,9 +300,51 @@ export interface CheckpointExecutionLoad {
239
300
  * so the /resume-plans brief can surface outstanding highs before the
240
301
  * per-turn injection ever runs. */
241
302
  findings?: ReviewFinding[];
303
+ /** v0.9.2: persisted blocker of the newest failed round, so the resume
304
+ * brief can name the open tasks that keep the review from starting. */
305
+ blocked?: ExecutionBlocked | null;
306
+ /** v0.9.3: the persisted per-run review budget (or undefined when the run
307
+ * never picked one — a legacy checkpoint resolves to 5 in the live state,
308
+ * see `reviewBudgetDefaulted`). */
309
+ reviewBudget?: ReviewBudget;
310
+ reviewBudgetDefaulted?: boolean;
311
+ reviewRoundsTotal?: number;
312
+ reviewCapExtension?: number;
242
313
  error?: string;
243
314
  }
244
315
 
316
+ /** v0.9.2: a checkpoint written before `execution.blocked` existed still names
317
+ * its blocker — `audit.lastResult` records the newest failed round's ids and
318
+ * the coverage cascade is pure, so the rollback set can be reconstructed
319
+ * WITHOUT re-applying anything (the tree already carries the reopen's outcome;
320
+ * re-applying it would revert tasks the executor has since re-closed).
321
+ * Finding-driven rounds (`highs: F-…`) carry no coverage and stay without a
322
+ * record — the wake still states that no round can start while tasks are open. */
323
+ function blockedFromFailedIds(
324
+ failedIds: readonly string[],
325
+ round: number,
326
+ tasks: TaskView[],
327
+ checklist: CheckItem[],
328
+ ): ExecBlocked | null {
329
+ if (failedIds.length === 0) return null;
330
+ const rolledBack = rollbackCoverageIds(tasks, checklist, failedIds, []);
331
+ if (rolledBack.length === 0) return null;
332
+ const open = blockedReviewTasks(tasks, rolledBack);
333
+ if (open.length === 0) return null;
334
+ return { rolledBack, tasks: open, round, escalatedRounds: 0, since: utcNow() };
335
+ }
336
+
337
+ function backfillBlocked(
338
+ checkpoint: ExecutionCheckpoint,
339
+ tasks: TaskView[],
340
+ checklist: CheckItem[],
341
+ ): ExecBlocked | null {
342
+ const last = checkpoint.audit?.lastResult?.trim() ?? "";
343
+ if (last.length === 0 || last.startsWith("highs:")) return null;
344
+ const failed = last.split(",").map((id) => id.trim()).filter((id) => id.length > 0);
345
+ return blockedFromFailedIds(failed, checkpoint.audit?.rounds ?? 0, tasks, checklist);
346
+ }
347
+
245
348
  /**
246
349
  * Shared restore primitive: load the executing state from a run checkpoint
247
350
  * into THIS session. Authorization is kept only when the recorded approval
@@ -309,6 +412,11 @@ export function loadExecutionFromCheckpoint(
309
412
  startedAt: utcNow(),
310
413
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
311
414
  uiLanguage: resolveUiLanguage(ctx.cwd),
415
+ // v0.9.2: the persisted blocker survives a restore so the resume brief
416
+ // and the dashboard can name it; a reverifyAll restore drops it together
417
+ // with the task progress it described. Pre-feature checkpoints are
418
+ // backfilled from the newest failed round's ids.
419
+ blocked: reverifyAll ? null : (cp.execution.blocked ?? backfillBlocked(cp.execution, tasks, items)),
312
420
  // D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
313
421
  // resume itself is the user's intent; the reason is surfaced in the
314
422
  // resume brief instead. EXCEPT a review-cap pause (v0.8): it must
@@ -327,8 +435,16 @@ export function loadExecutionFromCheckpoint(
327
435
  findings: toReviewFindings(cp.execution.audit?.findings),
328
436
  running: false,
329
437
  },
330
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
438
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
331
439
  auditLatch: { auditedThisSettle: false, activity: 0 },
440
+ // v0.9.3: the review budget survives a restore; a checkpoint written
441
+ // before the feature that already spent rounds keeps the legacy 5-round
442
+ // bound instead of being cut to the new default mid-flight.
443
+ reviewBudget: resolveStoredBudget(cp.execution.reviewBudget, cp.execution.audit?.rounds ?? 0),
444
+ reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted === true,
445
+ reviewRoundsTotal: cp.execution.reviewRoundsTotal ?? 0,
446
+ reviewCapExtension: cp.execution.reviewCapExtension ?? 0,
447
+ reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
332
448
  };
333
449
  execution.stall.lastSnapshot = stallSnapshot();
334
450
  executionRunId = runId;
@@ -363,6 +479,11 @@ export function loadExecutionFromCheckpoint(
363
479
  if (headChanged) {
364
480
  withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
365
481
  }
482
+ // A backfilled blocker is authoritative from here on: persist it once so a
483
+ // later restart reads the record instead of re-deriving it.
484
+ if (!reverifyAll && (cp.execution.blocked ?? null) === null && execution!.blocked !== null) {
485
+ withExecutionCheckpoint(ctx, (current) => applyExecutionProgress(current, { blocked: execution!.blocked }));
486
+ }
366
487
  persist(ctx);
367
488
  updateStatusWidget(ctx);
368
489
  return {
@@ -374,6 +495,13 @@ export function loadExecutionFromCheckpoint(
374
495
  legacyPlan: planTasks.legacy,
375
496
  legacyDelegate,
376
497
  findings: toReviewFindings(cp.execution.audit?.findings),
498
+ // The LIVE record: a pre-feature checkpoint is backfilled (and persisted)
499
+ // above, so the resume brief sees the reconstructed blocker.
500
+ blocked: execution?.blocked ?? null,
501
+ reviewBudget: execution?.reviewBudget,
502
+ reviewBudgetDefaulted: execution?.reviewBudgetDefaulted === true,
503
+ reviewRoundsTotal: execution?.reviewRoundsTotal ?? 0,
504
+ reviewCapExtension: execution?.reviewCapExtension ?? 0,
377
505
  };
378
506
  }
379
507
 
@@ -515,6 +643,8 @@ function updatePanelWidget(ctx: ExtensionContext): void {
515
643
  auditUndeterminable: current.audit.undeterminable,
516
644
  findings: current.audit.findings,
517
645
  reviewRunning: current.audit.running === true || current.review.inFlight !== null,
646
+ blockedTasks: blockedTaskIds(current),
647
+ blockedRound: current.blocked?.round ?? null,
518
648
  startedAt: current.startedAt,
519
649
  usage: current.usage,
520
650
  });
@@ -540,6 +670,12 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
540
670
  auditUndeterminable: execution.audit.undeterminable,
541
671
  findings: execution.audit.findings,
542
672
  reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
673
+ blockedTasks: blockedTaskIds(execution),
674
+ blockedRound: execution.blocked?.round ?? null,
675
+ reviewBudget: execution.reviewBudget ?? null,
676
+ reviewRoundsTotal: execution.reviewRoundsTotal,
677
+ reviewCapExtension: execution.reviewCapExtension,
678
+ reviewBudgetDefaulted: execution.reviewBudgetDefaulted === true,
543
679
  });
544
680
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
545
681
  return;
@@ -611,6 +747,15 @@ function persist(ctx: ExtensionContext): void {
611
747
  startedAt: execution.startedAt,
612
748
  usage: execution.usage,
613
749
  stall: execution.stall,
750
+ blocked: execution.blocked,
751
+ // v0.9.3 (round-1 F-005): the review budget rides the session snapshot
752
+ // too — a /reload restore rebuilds the live state from it, and a
753
+ // restored counter must NOT hand the run a free unlimited window.
754
+ reviewBudget: execution.reviewBudget,
755
+ reviewBudgetDefaulted: execution.reviewBudgetDefaulted,
756
+ reviewRoundsTotal: execution.reviewRoundsTotal,
757
+ reviewCapExtension: execution.reviewCapExtension,
758
+ reviewNoProgress: execution.reviewNoProgress,
614
759
  audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
615
760
  });
616
761
  }
@@ -654,9 +799,15 @@ export async function startExecution(
654
799
  usage: { inToks: 0, outToks: 0 },
655
800
  uiLanguage: resolveUiLanguage(ctx.cwd),
656
801
  stall: { rounds: 0, lastSnapshot: null, paused: false },
802
+ blocked: null,
657
803
  audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
658
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
804
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
659
805
  auditLatch: { auditedThisSettle: false, activity: 0 },
806
+ // v0.9.3: a fresh handoff starts with NO budget — the picker resolves it
807
+ // (or the no-UI default applies) right before round 1.
808
+ reviewBudget: undefined,
809
+ reviewRoundsTotal: 0,
810
+ reviewCapExtension: 0,
660
811
  };
661
812
  // Seed the watchdog baseline only after `execution` points at the new state
662
813
  // (stallSnapshot reads the live execution).
@@ -716,11 +867,21 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
716
867
  // Any task-state change resets the stall watchdog baseline.
717
868
  execution.stall.rounds = 0;
718
869
  execution.stall.lastSnapshot = stallSnapshot();
870
+ // v0.9.2: closing a reopened task shrinks the blocker; when the last one
871
+ // closes the record is dropped so no stale blocker outlives the repair.
872
+ if (execution.blocked) {
873
+ execution.blocked.tasks = blockedReviewTasks(execution.tasks, execution.blocked.rolledBack);
874
+ if (execution.blocked.tasks.length === 0) {
875
+ execution.blocked = null;
876
+ execution.blockedWakeTasks = undefined;
877
+ }
878
+ }
719
879
  withExecutionCheckpoint(ctx, (cp) =>
720
880
  applyExecutionProgress(cp, {
721
881
  tasks: taskProgressMap(execution!.tasks),
722
882
  doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
723
883
  stallRounds: execution!.stall.rounds,
884
+ blocked: execution!.blocked,
724
885
  audit: {
725
886
  rounds: execution!.audit.rounds,
726
887
  lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
@@ -884,14 +1045,102 @@ export function registerExecutionTurnHandlers(
884
1045
  *
885
1046
  * Completion stays fail-closed AND never fail-open: a run completes only
886
1047
  * when every pending check was affirmatively passed. */
887
- const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
888
- /** v0.7 protocol value — dual-matched for one release so checkpoints written
889
- * by older builds keep their cap pause recognized on restore. */
890
- const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
891
-
892
1048
  function isReviewCapPause(reason: string | undefined | null): boolean {
893
- if (!reason) return false;
894
- return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
1049
+ return isReviewPauseReason(reason);
1050
+ }
1051
+
1052
+ /** v0.9.3: the counters that bound an unlimited budget. */
1053
+ function budgetCounters(ex: ExecState): { reviewRoundsTotal: number; reviewCapExtension: number } {
1054
+ return { reviewRoundsTotal: ex.reviewRoundsTotal, reviewCapExtension: ex.reviewCapExtension };
1055
+ }
1056
+
1057
+ /** The budget in force for this run; a legacy checkpoint resolves to 5 during
1058
+ * load, so this is only a guard for hand-built test states. */
1059
+ function activeBudget(ex: ExecState): ReviewBudget {
1060
+ return ex.reviewBudget ?? LEGACY_REVIEW_MAX_ROUNDS;
1061
+ }
1062
+
1063
+ function budgetSpent(ex: ExecState): boolean {
1064
+ return budgetExhausted(activeBudget(ex), ex.audit.rounds, budgetCounters(ex));
1065
+ }
1066
+
1067
+ /** `3` / `∞` — the budget as shown in status lines and messages. */
1068
+ function budgetLabel(ex: ExecState): string {
1069
+ return formatReviewBudget(activeBudget(ex));
1070
+ }
1071
+
1072
+ /**
1073
+ * v0.9.3: whether the budget panel may be shown in THIS session. Beyond the
1074
+ * native-surface check, auto-approve sessions never ask — the recorded plan
1075
+ * decision routes headless/auto-approve runs to the default budget
1076
+ * (`PI_PLANS_AUTO_APPROVE=1` exists for unattended harnesses, where a panel
1077
+ * would either hang or answer meaninglessly).
1078
+ */
1079
+ function reviewBudgetPanelUsable(ctx: ExtensionContext): boolean {
1080
+ return !isAutoApproveEnabledLocal() && reviewBudgetPanelAvailable(ctx);
1081
+ }
1082
+
1083
+ /** Persist a freshly resolved budget (and the counters) in one revision. */
1084
+ function persistResolvedBudget(ctx: ExtensionContext, ex: ExecState): void {
1085
+ withExecutionCheckpoint(ctx, (cp) =>
1086
+ applyExecutionProgress(cp, {
1087
+ reviewBudget: ex.reviewBudget ?? null,
1088
+ reviewBudgetDefaulted: ex.reviewBudgetDefaulted === true,
1089
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1090
+ reviewCapExtension: ex.reviewCapExtension,
1091
+ }),
1092
+ );
1093
+ persist(ctx);
1094
+ updateStatusWidget(ctx);
1095
+ }
1096
+
1097
+ /** One visible note whenever the fallback (rather than a user pick) decided
1098
+ * the budget — never a silent bound (v0.9.3, Q-3). */
1099
+ function notifyDefaultBudget(ex: ExecState, why: "no-panel" | "cancelled"): void {
1100
+ const budget = formatReviewBudget(ex.reviewBudget ?? DEFAULT_REVIEW_BUDGET);
1101
+ messaging().sendMessage(
1102
+ {
1103
+ customType: "pi-plans-review-budget-default",
1104
+ content:
1105
+ why === "no-panel"
1106
+ ? `**pi-plans: execution review budget: ${budget} (default)** — no budget panel is available in this session, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`
1107
+ : `**pi-plans: execution review budget: ${budget} (default)** — the budget panel was closed without a choice, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`,
1108
+ display: true,
1109
+ },
1110
+ { triggerTurn: false },
1111
+ );
1112
+ }
1113
+
1114
+ /** Synchronous fallback: apply (and persist) the default budget. Used when no
1115
+ * panel exists at all — the headless/print/json and auto-approve paths — so
1116
+ * the first round needs no extra async hop. */
1117
+ function applyDefaultReviewBudget(ctx: ExtensionContext, ex: ExecState): void {
1118
+ if (ex.reviewBudget !== undefined) return;
1119
+ ex.reviewBudget = DEFAULT_REVIEW_BUDGET;
1120
+ ex.reviewBudgetDefaulted = true;
1121
+ notifyDefaultBudget(ex, "no-panel");
1122
+ persistResolvedBudget(ctx, ex);
1123
+ }
1124
+
1125
+ /**
1126
+ * Resolve the per-run review budget exactly once, immediately before the first
1127
+ * round. The panel path is async; every other case is handled up front by
1128
+ * `applyDefaultReviewBudget`. `budgetAsking` keeps a second settle (or a
1129
+ * restore while the panel is open) from opening a second panel.
1130
+ */
1131
+ async function askReviewBudgetForRun(ctx: ExtensionContext, ex: ExecState): Promise<void> {
1132
+ if (ex.reviewBudget !== undefined || ex.review.budgetAsking) return;
1133
+ ex.review.budgetAsking = true;
1134
+ try {
1135
+ const picked = await askReviewBudget(ctx, ex.uiLanguage, undefined);
1136
+ if (execution !== ex || ex.reviewBudget !== undefined) return;
1137
+ ex.reviewBudget = picked ?? DEFAULT_REVIEW_BUDGET;
1138
+ ex.reviewBudgetDefaulted = picked === null;
1139
+ if (picked === null) notifyDefaultBudget(ex, "cancelled");
1140
+ persistResolvedBudget(ctx, ex);
1141
+ } finally {
1142
+ ex.review.budgetAsking = false;
1143
+ }
895
1144
  }
896
1145
 
897
1146
  /** The currently-running (or self-scheduling) review chain; the sanctioned
@@ -933,8 +1182,56 @@ function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
933
1182
  * the review owed even when every check is done (the stranded-high path),
934
1183
  * mirroring how a failed check keeps it owed today. */
935
1184
  function reviewOwed(ex: ExecState): boolean {
936
- return allTasksTerminal(ex.tasks)
937
- && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
1185
+ return allTasksTerminal(ex.tasks) && reviewOutstanding(ex);
1186
+ }
1187
+
1188
+ /** v0.9.2: the review is outstanding wherever the tree stands — checks still
1189
+ * owed or an unresolved high finding. `reviewOwed` ANDs this with
1190
+ * allTasksTerminal; the blocked wake needs exactly the half that stays true
1191
+ * while tasks are open, because that is the state in which the review cannot
1192
+ * start (and in which `pendingAudit` is false by construction). */
1193
+ function reviewOutstanding(ex: ExecState): boolean {
1194
+ return auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0;
1195
+ }
1196
+
1197
+ /** v0.9.2: live blocker ids — the newest failed round's rollback set
1198
+ * intersected with the still-open tasks. Always recomputed from the tree: the
1199
+ * stored `tasks` list may predate the last status change. */
1200
+ function blockedTaskIds(ex: ExecState): string[] {
1201
+ return ex.blocked ? blockedReviewTasks(ex.tasks, ex.blocked.rolledBack) : [];
1202
+ }
1203
+
1204
+ /** v0.9.2: one visible system line per escalation level, so the user sees the
1205
+ * loop escalating before the watchdog pauses it. Memory-only latch (the
1206
+ * restored state re-notifies once, which is the useful behavior). */
1207
+ function notifyBlockedEscalation(ctx: ExtensionContext, ex: ExecState): void {
1208
+ const level = ex.blocked?.escalatedRounds ?? 0;
1209
+ if (!ex.blocked || level === 0 || ex.blockedNotifiedLevel === level) return;
1210
+ ex.blockedNotifiedLevel = level;
1211
+ const ids = blockedTaskIds(ex);
1212
+ try {
1213
+ messaging().sendMessage(
1214
+ {
1215
+ customType: EXECUTION_BLOCKED_CUSTOM_TYPE,
1216
+ content: `**pi-plans: execution blocked (wake ${level}/${STALL_MAX_ROUNDS})** — review round ${ex.blocked.round + 1} cannot start: ${ids.join(", ")} ${ids.length === 1 ? "was" : "were"} reopened by round ${ex.blocked.round} and ${ids.length === 1 ? "is" : "are"} still open. Close them with \`plans_update_task\` (the review starts by itself once every task is terminal).`,
1217
+ display: true,
1218
+ },
1219
+ { triggerTurn: false },
1220
+ );
1221
+ } catch {
1222
+ /* best-effort: the wake below still carries the same blocker */
1223
+ }
1224
+ }
1225
+
1226
+ /** v0.9.2: the pause reason names the real blocker instead of the watchdog's
1227
+ * own metric. Single line, never a review-cap prefix (`isReviewCapPause`
1228
+ * matches "execution review exhausted" / "completion audit exhausted"). */
1229
+ function blockedPauseReason(ex: ExecState): string {
1230
+ const ids = blockedTaskIds(ex);
1231
+ const round = ex.blocked?.round ?? ex.audit.rounds;
1232
+ const head = ids.slice(0, 3).join(", ");
1233
+ const more = ids.length > 3 ? `, +${ids.length - 3} more` : "";
1234
+ return `blocked: review round ${round + 1} cannot start — ${ids.length} task(s) reopened by round ${round} still open (${head}${more})`;
938
1235
  }
939
1236
 
940
1237
  function runDirOf(ctx: ExtensionContext): string | null {
@@ -1076,7 +1373,13 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1076
1373
  : highs.length > 0
1077
1374
  ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1078
1375
  : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1079
- const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1376
+ // v0.9.3: the reason names the budget actually in force — the numeric round
1377
+ // count, or the unlimited hard cap that stopped the loop.
1378
+ const scope =
1379
+ activeBudget(ex) === "unlimited"
1380
+ ? `${unlimitedHardCapCeiling(budgetCounters(ex))} rounds (unlimited budget safety cap)`
1381
+ : `${activeBudget(ex)} round${activeBudget(ex) === 1 ? "" : "s"}`;
1382
+ const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${scope} (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the ${formatReviewBudget(activeBudget(ex))} picker; ordinary messages and session restores do not). (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1080
1383
  pauseForStall(ctx, reason);
1081
1384
  // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
1082
1385
  // pauseForStall's ui.notify is optional and absent headless, so the pause
@@ -1087,6 +1390,41 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1087
1390
  );
1088
1391
  }
1089
1392
 
1393
+ /** v0.9.3: the unlimited budget's no-progress valve — three consecutive
1394
+ * committed rounds with an identical outcome signature cannot converge, so
1395
+ * the run pauses fail-closed instead of burning rounds silently. */
1396
+ function pauseReviewNoProgress(ctx: ExtensionContext, ex: ExecState): void {
1397
+ const streak = ex.reviewNoProgress?.streak ?? NO_PROGRESS_MAX_STREAK;
1398
+ const highs = unresolvedHighFindings(ex);
1399
+ const detail =
1400
+ ex.audit.failed.length > 0
1401
+ ? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
1402
+ : highs.length > 0
1403
+ ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1404
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1405
+ const reason = `${REVIEW_NO_PROGRESS_PAUSE_PREFIX} — ${streak} consecutive rounds reported the same outcome (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the budget picker; ordinary messages and session restores do not).`;
1406
+ pauseForStall(ctx, reason);
1407
+ messaging().sendMessage(
1408
+ { customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
1409
+ { triggerTurn: false },
1410
+ );
1411
+ }
1412
+
1413
+ /** v0.9.3: the single gate every round spawn passes through. Returns false
1414
+ * when the run was paused (no round may start). The unlimited valve order is
1415
+ * deliberate: no-progress is checked first so its reason wins when both hold. */
1416
+ function reviewBudgetGate(ctx: ExtensionContext, ex: ExecState): boolean {
1417
+ if (activeBudget(ex) === "unlimited" && noProgressTripped(ex.reviewNoProgress)) {
1418
+ pauseReviewNoProgress(ctx, ex);
1419
+ return false;
1420
+ }
1421
+ if (budgetSpent(ex)) {
1422
+ pauseReviewCap(ctx, ex);
1423
+ return false;
1424
+ }
1425
+ return true;
1426
+ }
1427
+
1090
1428
  async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1091
1429
  if (!execution) return;
1092
1430
  const ex = execution;
@@ -1105,11 +1443,27 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1105
1443
  await completeExecution(ctx);
1106
1444
  return;
1107
1445
  }
1108
- if (ex.stall.paused || ex.review.inFlight) return;
1109
- if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1110
- pauseReviewCap(ctx, ex);
1111
- return;
1446
+ if (ex.stall.paused || ex.review.inFlight || ex.review.budgetAsking) return;
1447
+ // v0.9.3: the budget is resolved right here — every task is terminal, the
1448
+ // review is genuinely owed, and this is the last moment before round 1.
1449
+ // No panel → the default applies synchronously (no extra tick, so a
1450
+ // detached settle still shows the round in flight immediately); a panel →
1451
+ // one await, guarded against a second settle by `budgetAsking`.
1452
+ if (ex.reviewBudget === undefined && !reviewBudgetPanelUsable(ctx)) applyDefaultReviewBudget(ctx, ex);
1453
+ if (ex.reviewBudget === undefined) {
1454
+ await askReviewBudgetForRun(ctx, ex);
1455
+ if (execution !== ex || ex.reviewBudget === undefined) return;
1456
+ }
1457
+ if (pendingChecks.length === 0) {
1458
+ // Every verification check is satisfied; only unresolved highs keep the
1459
+ // review owed. With the budget spent there is no round left to buy, so
1460
+ // the run completes and DISCLOSES the highs (v0.9.3, Q-2).
1461
+ if (budgetSpent(ex)) {
1462
+ await completeExecution(ctx);
1463
+ return;
1464
+ }
1112
1465
  }
1466
+ if (!reviewBudgetGate(ctx, ex)) return;
1113
1467
  // Phase transition: the executor is done with its tasks; the review loop
1114
1468
  // owns the run until it converges (or pauses at the cap).
1115
1469
  setRunStatusForReview(ctx, "verifying");
@@ -1320,6 +1674,9 @@ async function commitReviewOutcome(
1320
1674
  // The budget is charged only when an outcome commits — never on discard
1321
1675
  // or cancellation (CF2-003).
1322
1676
  ex.audit.rounds = round.budgetRound;
1677
+ // v0.9.3: the run-cumulative counter never resets — it is what bounds an
1678
+ // unlimited budget across grants (Q-4).
1679
+ ex.reviewRoundsTotal += 1;
1323
1680
  const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
1324
1681
  const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
1325
1682
  // Anything the round neither passed nor failed is undeterminable: the
@@ -1337,6 +1694,12 @@ async function commitReviewOutcome(
1337
1694
  const findings = reported ? outcome.findings : ex.audit.findings;
1338
1695
  ex.audit.findings = findings;
1339
1696
  const highs = findings.filter((f) => f.severity === "high");
1697
+ // v0.9.3: the no-progress valve's signature is computed HERE, from the
1698
+ // post-classification triple — a spawn-failure round (`outcome === null`)
1699
+ // and a discard synthesis have no findings array of their own, but the
1700
+ // preserved unresolved set plus the derived verdicts still describe a
1701
+ // concrete, comparable outcome (round-1 F-002).
1702
+ ex.reviewNoProgress = bumpNoProgress(ex.reviewNoProgress, noProgressSignature(failed, undeterminable, highs.map((f) => f.id)));
1340
1703
  // Findings are actionable only when this round actually reported them;
1341
1704
  // see the fix-loop branch below (v0.9.1, F-001).
1342
1705
  const actionableHighs = reported ? highs : [];
@@ -1371,6 +1734,40 @@ async function commitReviewOutcome(
1371
1734
  if (item) item.done = true;
1372
1735
  }
1373
1736
 
1737
+ // v0.9.3 (Q-2/F-003): the exhausted-budget completion. EVERY check this
1738
+ // round still owed affirmed, only unresolved highs remain, and the budget
1739
+ // cannot buy another round — the run completes, tolerating the highs (they
1740
+ // stay in the round report and the checkpoint; `completeExecution`
1741
+ // discloses them). Evaluated BEFORE the fix branch on purpose: the
1742
+ // tolerated round must not roll back tasks, append plan tasks, rewrite the
1743
+ // approved plan, or wake the executor for a run that is about to be done.
1744
+ if (budgetSpent(ex) && passed.length === pendingIds.length && highs.length > 0) {
1745
+ ex.audit.failed = [];
1746
+ ex.audit.undeterminable = [];
1747
+ ex.blocked = null;
1748
+ withExecutionCheckpoint(ctx, (cp) =>
1749
+ applyExecutionProgress(cp, {
1750
+ tasks: taskProgressMap(ex.tasks),
1751
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1752
+ blocked: null,
1753
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1754
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1755
+ audit: { rounds: ex.audit.rounds, passed: true, findings },
1756
+ }),
1757
+ );
1758
+ persist(ctx);
1759
+ messaging().sendMessage(
1760
+ {
1761
+ customType: "pi-plans-review-tolerated",
1762
+ content: `**pi-plans: review budget exhausted (${budgetLabel(ex)}) — completing with ${highs.length} unresolved high finding(s): ${highs.map((f) => f.id).join(", ")}** — every verification check passed; the finding(s) remain in the round reports.`,
1763
+ display: true,
1764
+ },
1765
+ { triggerTurn: false },
1766
+ );
1767
+ await completeExecution(ctx);
1768
+ return;
1769
+ }
1770
+
1374
1771
  // v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
1375
1772
  // high finding can never complete the run (liveness). Failed checks and
1376
1773
  // high findings drive ONE union rollback and exactly one executor wake;
@@ -1398,6 +1795,20 @@ async function commitReviewOutcome(
1398
1795
  const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
1399
1796
  const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
1400
1797
  const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
1798
+ // v0.9.2: capture the round's rollback set HERE. The reopen helpers above
1799
+ // mutate the tree and report only flipped nodes, so this is the only
1800
+ // moment the authoritative provenance exists; the still-open subset is
1801
+ // what blocks the next round and what wakes/pauses name explicitly.
1802
+ ex.blocked = allRolledBack.length > 0
1803
+ ? {
1804
+ rolledBack: [...allRolledBack],
1805
+ tasks: blockedReviewTasks(ex.tasks, allRolledBack),
1806
+ round: ex.audit.rounds,
1807
+ escalatedRounds: 0,
1808
+ since: utcNow(),
1809
+ }
1810
+ : null;
1811
+ ex.blockedWakeTasks = undefined;
1401
1812
  withExecutionCheckpoint(ctx, (cp) => {
1402
1813
  // v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
1403
1814
  // re-stamp the checkpoint's plan identity in the same revision so a
@@ -1409,6 +1820,9 @@ async function commitReviewOutcome(
1409
1820
  return applyExecutionProgress(amendedCp, {
1410
1821
  tasks: taskProgressMap(ex.tasks),
1411
1822
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1823
+ blocked: ex.blocked,
1824
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1825
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1412
1826
  audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
1413
1827
  });
1414
1828
  });
@@ -1442,7 +1856,7 @@ async function commitReviewOutcome(
1442
1856
  ? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
1443
1857
  : `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
1444
1858
  const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
1445
- const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
1859
+ const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${budgetSpent(ex) ? (activeBudget(ex) === "unlimited" ? ` This was round ${ex.reviewRoundsTotal} against the unlimited budget's ${unlimitedHardCapCeiling(budgetCounters(ex))}-round safety cap: the next terminal-task cycle pauses the run for review.` : ` This was round ${ex.audit.rounds} of ${budgetLabel(ex)}: the next terminal-task cycle pauses the run for review (or completes if every check passed).`) : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
1446
1860
  messaging().sendMessage(
1447
1861
  {
1448
1862
  customType: "pi-plans-audit-failed",
@@ -1465,10 +1879,15 @@ async function commitReviewOutcome(
1465
1879
  if (passed.length === pendingIds.length && highs.length === 0) {
1466
1880
  ex.audit.failed = [];
1467
1881
  ex.audit.undeterminable = [];
1882
+ // v0.9.2: a passing audit ends the blocker — the review consumed it.
1883
+ ex.blocked = null;
1468
1884
  withExecutionCheckpoint(ctx, (cp) =>
1469
1885
  applyExecutionProgress(cp, {
1470
1886
  tasks: taskProgressMap(ex.tasks),
1471
1887
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1888
+ blocked: null,
1889
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1890
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1472
1891
  audit: { rounds: ex.audit.rounds, passed: true, findings },
1473
1892
  }),
1474
1893
  );
@@ -1484,6 +1903,8 @@ async function commitReviewOutcome(
1484
1903
  applyExecutionProgress(cp, {
1485
1904
  tasks: taskProgressMap(ex.tasks),
1486
1905
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1906
+ reviewRoundsTotal: ex.reviewRoundsTotal,
1907
+ reviewNoProgress: ex.reviewNoProgress ?? null,
1487
1908
  audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
1488
1909
  }),
1489
1910
  );
@@ -1496,10 +1917,11 @@ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promis
1496
1917
  if (execution !== ex) return;
1497
1918
  if (!reviewOwed(ex)) return;
1498
1919
  if (ex.stall.paused) return;
1499
- if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1500
- pauseReviewCap(ctx, ex);
1501
- return;
1502
- }
1920
+ // The budget gate owns every pause (numeric exhaustion, the unlimited hard
1921
+ // cap, and the unlimited no-progress valve). An exhausted budget with every
1922
+ // check satisfied has already completed inside commitReviewOutcome; here
1923
+ // the review is still owed, so a spent budget pauses.
1924
+ if (!reviewBudgetGate(ctx, ex)) return;
1503
1925
  await startReviewRound(ctx);
1504
1926
  }
1505
1927
 
@@ -2085,6 +2507,8 @@ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
2085
2507
  // This also bounds the zero-input continue loop in agent_before_settle.
2086
2508
  if (ex.stall.paused) return false;
2087
2509
  if (ex.review.inFlight) return false;
2510
+ // The budget panel is open: the review is being resolved, not owed anew.
2511
+ if (ex.review.budgetAsking) return false;
2088
2512
  return allTasksTerminal(ex.tasks)
2089
2513
  && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
2090
2514
  }
@@ -2171,6 +2595,33 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
2171
2595
  if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
2172
2596
  runtime.handled = true;
2173
2597
  const ex = runtime.owner;
2598
+ // v0.9.2: while a failed round's tasks are still open, the next round
2599
+ // cannot start — and this ladder must NOT ride `stall.rounds`, which any
2600
+ // successful tool call resets: an executor that investigates but never
2601
+ // closes the reopened tasks has to escalate (and eventually pause) anyway.
2602
+ const blockedIds = blockedTaskIds(ex);
2603
+ if (ex.blocked && blockedIds.length > 0) {
2604
+ // Progress = fewer blockers than at the PREVIOUS blocked wake. The
2605
+ // stored `blocked.tasks` cannot serve as that baseline: every task
2606
+ // close re-syncs it, so it always equals the live set and the ladder
2607
+ // would only ever increment (review round 1, F-001). A same-size but
2608
+ // different set counts as no progress; a new member counts as none.
2609
+ const baseline = ex.blockedWakeTasks;
2610
+ const progressed = baseline !== undefined && blockedIds.length < baseline.length;
2611
+ ex.blockedWakeTasks = [...blockedIds];
2612
+ ex.blocked.escalatedRounds = progressed ? 0 : ex.blocked.escalatedRounds + 1;
2613
+ ex.blocked.tasks = blockedIds;
2614
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { blocked: ex.blocked }));
2615
+ if (ex.blocked.escalatedRounds >= STALL_MAX_ROUNDS) {
2616
+ pauseForStall(ctx, blockedPauseReason(ex));
2617
+ return;
2618
+ }
2619
+ notifyBlockedEscalation(ctx, ex);
2620
+ persist(ctx);
2621
+ updateStatusWidget(ctx);
2622
+ sendContinuationWake(ctx, runtime);
2623
+ return;
2624
+ }
2174
2625
  const snapshot = stallSnapshot();
2175
2626
  const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
2176
2627
  ex.stall.lastSnapshot = snapshot;
@@ -2190,8 +2641,10 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
2190
2641
 
2191
2642
  export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
2192
2643
  // v0.6.1: continuation wakes are one-shot; stale ones (including the
2193
- // legacy v0.6.0 goal-wait type) never replay after a restart.
2644
+ // legacy v0.6.0 goal-wait type) never replay after a restart. v0.9.2: the
2645
+ // visible escalated-blocked line is one-shot for the same reason.
2194
2646
  return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
2647
+ && message.customType !== EXECUTION_BLOCKED_CUSTOM_TYPE
2195
2648
  && message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
2196
2649
  }
2197
2650
 
@@ -2200,18 +2653,20 @@ export function filterContinuationMessages<T extends { customType?: string; deta
2200
2653
  }
2201
2654
 
2202
2655
  /** Called for genuine user input or an explicit same-execution resume.
2203
- * v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
2204
- * refill the five-round budget (CF2-004); only /plans-execute
2656
+ * v0.8: a REVIEW pause is never lifted here — ordinary input must not
2657
+ * refill the review budget (CF2-004); only /plans-execute
2205
2658
  * (resumeActiveExecution) is the explicit confirmation surface. Genuine
2206
2659
  * stall pauses still clear on input as before. */
2207
2660
  export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
2208
2661
  const ex = getExecution();
2209
2662
  if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
2210
- if (isReviewCapPause(ex.stall.pausedReason)) {
2663
+ if (isReviewPauseReason(ex.stall.pausedReason)) {
2211
2664
  // Surfaced once per input so the user is not left guessing why the run
2212
2665
  // stays paused; the pause itself and the budget survive untouched.
2666
+ // v0.9.3: the note names the budget actually in force and the fact that
2667
+ // the confirmation re-opens the picker.
2213
2668
  ctx.ui.notify?.(
2214
- "pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
2669
+ `pi-plans: the review budget is exhausted (${budgetLabel(ex)}) — run /plans-execute to grant a fresh budget (it re-opens the round-count picker; that confirmation is the only surface that does).`,
2215
2670
  "warning",
2216
2671
  );
2217
2672
  return false;
@@ -2220,18 +2675,61 @@ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
2220
2675
  ex.stall.pausedReason = undefined;
2221
2676
  ex.stall.rounds = 0;
2222
2677
  ex.stall.lastSnapshot = stallSnapshot();
2223
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
2678
+ // v0.9.2: a resume grants a fresh escalation ladder (the blocker itself is
2679
+ // still recorded, so the next wake names it) and never a fresh review
2680
+ // budget — that stays `/plans-execute`-only.
2681
+ if (ex.blocked) {
2682
+ ex.blocked.escalatedRounds = 0;
2683
+ ex.blockedWakeTasks = undefined;
2684
+ }
2685
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null, blocked: ex.blocked }));
2224
2686
  persist(ctx);
2225
2687
  updateStatusWidget(ctx);
2226
2688
  return true;
2227
2689
  }
2228
2690
 
2229
- export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2230
- // v0.8: /plans-execute is THE explicit confirmation surface for a
2231
- // review-cap pause — the only place a fresh five-round budget is granted
2232
- // (Q-confirm-surface). Ordinary input and session restores never refill.
2691
+ /** Outcome of an explicit `/plans-execute` resume, so the calling tool can
2692
+ * report what actually happened (v0.9.3: the grant may re-open the picker). */
2693
+ export interface ResumeOutcome {
2694
+ resumed: boolean;
2695
+ /** Set when the pause was lifted by a budget grant. */
2696
+ grantedBudget?: ReviewBudget;
2697
+ /** True when the user closed the budget panel — the pause stands. */
2698
+ budgetDeclined?: boolean;
2699
+ }
2700
+
2701
+ export async function resumeActiveExecution(ctx: ExtensionContext): Promise<ResumeOutcome> {
2702
+ // v0.8: /plans-execute is THE explicit confirmation surface for a review
2703
+ // pause — the only place a fresh budget is granted (Q-confirm-surface).
2704
+ // Ordinary input and session restores never refill.
2705
+ // v0.9.3: the grant re-opens the budget picker with the current value
2706
+ // preselected; Esc keeps the run paused (an explicit confirmation is the
2707
+ // only way forward). Headless sessions, which have no panel to show, keep
2708
+ // the current budget instead of stranding the run.
2233
2709
  const pausedEx = getExecution();
2234
- if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
2710
+ if (pausedEx?.stall.paused && isReviewPauseReason(pausedEx.stall.pausedReason)) {
2711
+ const previous = activeBudget(pausedEx);
2712
+ let granted = previous;
2713
+ if (reviewBudgetPanelUsable(ctx)) {
2714
+ const picked = await askReviewBudget(ctx, pausedEx.uiLanguage, previous);
2715
+ if (picked === null) {
2716
+ messaging().sendMessage(
2717
+ {
2718
+ customType: "pi-plans-review-budget-declined",
2719
+ content: `**pi-plans: review budget unchanged (${formatReviewBudget(previous)})** — the run stays paused. Run /plans-execute and pick a round count to continue.`,
2720
+ display: true,
2721
+ },
2722
+ { triggerTurn: false },
2723
+ );
2724
+ return { resumed: false, budgetDeclined: true };
2725
+ }
2726
+ granted = picked;
2727
+ // Q-4: only a grant that lands on `unlimited` lifts the hard cap —
2728
+ // the cumulative counter itself never resets.
2729
+ if (granted === "unlimited") {
2730
+ pausedEx.reviewCapExtension += UNLIMITED_HARD_CAP;
2731
+ }
2732
+ }
2235
2733
  pausedEx.stall.paused = false;
2236
2734
  pausedEx.stall.pausedReason = undefined;
2237
2735
  pausedEx.stall.rounds = 0;
@@ -2239,28 +2737,45 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2239
2737
  pausedEx.audit.rounds = 0;
2240
2738
  pausedEx.audit.failed = [];
2241
2739
  pausedEx.audit.undeterminable = [];
2740
+ // v0.9.3: a fresh budget window also resets the no-progress valve.
2741
+ pausedEx.reviewNoProgress = undefined;
2742
+ // v0.9.2: the explicit confirmation also refreshes the blocked ladder
2743
+ // (the blocker set itself survives — it still names what stays open).
2744
+ if (pausedEx.blocked) {
2745
+ pausedEx.blocked.escalatedRounds = 0;
2746
+ pausedEx.blockedWakeTasks = undefined;
2747
+ }
2242
2748
  // v0.9: the fresh budget inherits unresolved findings (stable ids keep
2243
2749
  // counting) — only the round counter resets.
2244
2750
  withExecutionCheckpoint(ctx, (cp) =>
2245
2751
  applyExecutionProgress(cp, {
2246
2752
  tasks: taskProgressMap(pausedEx.tasks),
2247
2753
  audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
2754
+ blocked: pausedEx.blocked,
2755
+ reviewBudget: granted,
2756
+ reviewBudgetDefaulted: pausedEx.reviewBudgetDefaulted === true,
2757
+ reviewRoundsTotal: pausedEx.reviewRoundsTotal,
2758
+ reviewCapExtension: pausedEx.reviewCapExtension,
2759
+ reviewNoProgress: null,
2248
2760
  pausedReason: null,
2249
2761
  }),
2250
2762
  );
2251
2763
  persist(ctx);
2252
2764
  updateStatusWidget(ctx);
2765
+ const grantedNote = granted === "unlimited"
2766
+ ? `unlimited review budget granted (hard cap now ${unlimitedHardCapCeiling(budgetCounters(pausedEx))} rounds; the run has spent ${pausedEx.reviewRoundsTotal})`
2767
+ : `fresh ${formatReviewBudget(granted)}-round review budget granted`;
2253
2768
  messaging().sendMessage(
2254
2769
  {
2255
2770
  customType: "pi-plans-review-budget-granted",
2256
- content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
2771
+ content: `**pi-plans: ${grantedNote}** — the execution review resumes now.`,
2257
2772
  display: true,
2258
2773
  },
2259
2774
  { triggerTurn: false },
2260
2775
  );
2261
2776
  const grantChain = launchReviewRound(ctx);
2262
2777
  if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
2263
- return true;
2778
+ return { resumed: true, grantedBudget: granted };
2264
2779
  }
2265
2780
  // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
2266
2781
  // to `return false` here, so `/plans-execute` answered "already executing"
@@ -2271,15 +2786,15 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2271
2786
  latchAuditThisSettle();
2272
2787
  const chain = launchReviewRound(ctx);
2273
2788
  if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
2274
- return true;
2789
+ return { resumed: true };
2275
2790
  }
2276
- if (!resumeGoalWaitIfPaused(ctx)) return false;
2791
+ if (!resumeGoalWaitIfPaused(ctx)) return { resumed: false };
2277
2792
  const runtime = currentContinuationRuntime(ctx)!;
2278
2793
  if (canWakeExecution(ctx, runtime)) {
2279
2794
  runtime.handled = true;
2280
2795
  sendContinuationWake(ctx, runtime);
2281
2796
  }
2282
- return true;
2797
+ return { resumed: true };
2283
2798
  }
2284
2799
 
2285
2800
  export async function completeExecution(ctx: ExtensionContext): Promise<void> {
@@ -2297,6 +2812,13 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2297
2812
  const residualNote = residualFindings.length > 0
2298
2813
  ? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
2299
2814
  : "";
2815
+ // v0.9.3 (Q-2, round-1 F-004): an exhausted budget may complete WITH
2816
+ // unresolved high findings — the completion surface must say so instead of
2817
+ // claiming "execution review passed".
2818
+ const toleratedHighs = execution.audit.findings.filter((f) => f.severity === "high");
2819
+ const toleratedNote = toleratedHighs.length > 0
2820
+ ? `\n\n⚠️ The review budget was exhausted (${budgetLabel(execution)}) before these high-severity finding(s) could be resolved: ${toleratedHighs.map((f) => f.id).join(", ")} — every verification check passed; the finding(s) remain in the round reports under the run directory.`
2821
+ : "";
2300
2822
  const planPath = execution.planPath;
2301
2823
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
2302
2824
  execution = null;
@@ -2306,7 +2828,9 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2306
2828
  messaging().sendMessage(
2307
2829
  {
2308
2830
  customType: "pi-plans-complete",
2309
- content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
2831
+ content: toleratedHighs.length > 0
2832
+ ? `**Plan complete (review budget exhausted).** ⚠️ \`${planPath}\` — all verification checks passed; ${toleratedHighs.length} high-severity finding(s) stayed unresolved.\n\n${summary}${residualNote}${toleratedNote}`
2833
+ : `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
2310
2834
  display: true,
2311
2835
  },
2312
2836
  { triggerTurn: false },
@@ -2345,13 +2869,28 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
2345
2869
  const highFindingsNote = unresolvedHighs.length > 0
2346
2870
  ? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
2347
2871
  : "";
2872
+ // v0.9.2: the wake must answer "is the review running?" explicitly. The
2873
+ // previous text left the executor waiting for a round that cannot start
2874
+ // while the tree is open (the exact stall this feature fixes).
2875
+ const blockedIds = blockedTaskIds(execution);
2876
+ const blockedRound = execution.blocked?.round ?? execution.audit.rounds;
2877
+ const outstanding = reviewOutstanding(execution);
2878
+ const reviewLine = execution.review.inFlight
2879
+ ? `\nReview: round ${execution.audit.rounds + 1}/${budgetLabel(execution)} IS RUNNING (read-only reviewer verifying) — do not wait on it and do not re-close tasks for it.`
2880
+ : blockedIds.length > 0 && outstanding
2881
+ ? `\nReview: NO round is running — the task tree is not terminal, so the review cannot start. It starts by itself the moment every task is terminal (round ${execution.audit.rounds + 1}/${budgetLabel(execution)}).`
2882
+ : "";
2883
+ const escalated = (execution.blocked?.escalatedRounds ?? 0) > 0;
2884
+ const blockedLine = blockedIds.length > 0 && outstanding
2885
+ ? `\nBLOCKED — ${blockedIds.length} task(s) reopened by round ${blockedRound} are still open:\n${blockedIds.map((id) => `- ${id} (reopened by round ${blockedRound}) — close with \`plans_update_task\`: status "complete" with evidence, or "skipped" with a skipReason. Closing a child does NOT close its parent; a parent with an open child is not terminal.`).join("\n")}${escalated ? `\nThis is blocked wake ${execution.blocked?.escalatedRounds}/${STALL_MAX_ROUNDS}: if these tasks stay open, the watchdog pauses the run and a human has to resume it.` : ""}`
2886
+ : "";
2348
2887
  return `[PI-PLANS EXECUTION — write access enabled]
2349
2888
  Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
2350
2889
 
2351
2890
  Current wave ${currentWave} open tasks:
2352
2891
  ${waveList}
2353
2892
 
2354
- Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}
2893
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}${reviewLine}${blockedLine}
2355
2894
 
2356
2895
  ${graphLine}
2357
2896
 
@@ -2359,6 +2898,8 @@ Execution rules:
2359
2898
  - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
2360
2899
  - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
2361
2900
  - Close subtasks before their parent; a parent is auditable only when every child is terminal.
2901
+ - After a FAILED review round, every reopened task must be re-closed with fresh evidence — PARENTS INCLUDED. Closing a task's children does NOT close the task: a parent that still has an open child is not terminal, and the review cannot start until the whole tree is terminal.
2902
+ - NEVER wait for the review. While any task is open, the review is not running and nothing will re-close tasks for you: an open task is your work queue — close it with evidence or skip it with a skipReason.
2362
2903
  - When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
2363
2904
  - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
2364
2905
  - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
@@ -2438,6 +2979,10 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2438
2979
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
2439
2980
  uiLanguage: resolveUiLanguage(ctx.cwd),
2440
2981
  stall: { ...snapshot.stall, lastSnapshot: null },
2982
+ // v0.9.2: a snapshot taken by a pre-feature build carries the live failed
2983
+ // set but no blocker record; reconstruct it so the very first wake after
2984
+ // a /reload (the recovery path for an already-stuck run) names the tasks.
2985
+ blocked: snapshot.blocked ?? blockedFromFailedIds(snapshot.audit?.failed ?? [], snapshot.audit?.rounds ?? 0, tasks, items),
2441
2986
  audit: {
2442
2987
  rounds: snapshot.audit?.rounds ?? 0,
2443
2988
  failed: snapshot.audit?.failed ?? [],
@@ -2445,8 +2990,16 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2445
2990
  findings: toReviewFindings(snapshot.audit?.findings),
2446
2991
  running: false,
2447
2992
  },
2448
- review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
2993
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
2449
2994
  auditLatch: { auditedThisSettle: false, activity: 0 },
2995
+ // v0.9.3 (round-1 F-005): the snapshot's budget/counters survive a
2996
+ // /reload — an unset budget stays unset (the picker runs before round 1),
2997
+ // a legacy snapshot resolves to the 5-round bound via the audit counter.
2998
+ reviewBudget: resolveStoredBudget(snapshot.reviewBudget, snapshot.audit?.rounds ?? 0),
2999
+ reviewBudgetDefaulted: snapshot.reviewBudgetDefaulted === true,
3000
+ reviewRoundsTotal: snapshot.reviewRoundsTotal ?? 0,
3001
+ reviewCapExtension: snapshot.reviewCapExtension ?? 0,
3002
+ reviewNoProgress: snapshot.reviewNoProgress ?? undefined,
2450
3003
  };
2451
3004
  execution.stall.lastSnapshot = stallSnapshot();
2452
3005
  resetContinuationRuntime(ctx);