pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
package/src/exec.ts CHANGED
@@ -1,15 +1,24 @@
1
1
  /**
2
- * Plan-execution loop: the tracked execution mode for accepted plans.
2
+ * Plan-execution loop (v0.6.1): the tracked execution mode for accepted
3
+ * plans, driven by the plan's task tree.
3
4
  *
4
5
  * When the user approves the execution handoff, the extension switches into
5
- * execution mode: every agent turn is injected with the remaining verifier
6
- * checklist, assistant messages are scanned for [DONE:VC-xxx] markers, and
7
- * progress is reported through the bottom status bar until every item passes.
6
+ * execution mode: every agent turn is injected with the current wave and
7
+ * remaining tasks, task progress flows in exclusively through the
8
+ * `plans_update_task` tool (status + evidence), the task dashboard shows
9
+ * live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
10
+ * tree), a stall watchdog pauses the run when consecutive rounds produce no
11
+ * task-state change, and when every task reaches a terminal state an
12
+ * independent execution reviewer verifies the plan's verification checks in
13
+ * a detached, overlay-visible loop — failed checks roll their covered tasks
14
+ * back to pending (audit-flow-only channel), and five committed rounds pause
15
+ * the run for the user in every mode (fail-closed, never a silent stop).
8
16
  */
9
17
 
10
18
  import * as fs from "node:fs";
11
19
  import * as path from "node:path";
12
- import { randomUUID } from "node:crypto";
20
+ import { createHash, randomUUID } from "node:crypto";
21
+ import { execSync } from "node:child_process";
13
22
  import type {
14
23
  CompactionResult,
15
24
  ExtensionAPI,
@@ -36,11 +45,10 @@ import {
36
45
  type VccCompactionBuildResult,
37
46
  type VccCompactionStats,
38
47
  } from "./compaction.ts";
39
- import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
40
- import { execChrome, resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
48
+ import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, runDirPath, setRunStatus, StateError, utcNow } from "./state.ts";
49
+ import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
41
50
  import { bindRun, resolveActiveRun } from "./run-context.ts";
42
- import { runPiSubagent, type SubagentProgressEvent } from "./subagent.ts";
43
- import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
51
+ import type { SubagentProgressEvent } from "./subagent.ts";
44
52
  import { OwnershipError } from "./run-ownership.ts";
45
53
  import {
46
54
  applyExecutionApproved,
@@ -48,86 +56,104 @@ import {
48
56
  applyExecutionHeadChanged,
49
57
  applyExecutionProgress,
50
58
  applyExecutionStopped,
51
- applyQuestionAsked,
52
- applyReviewRoundStarted,
53
59
  createCheckpoint,
54
60
  loadCheckpoint,
55
61
  mutateCheckpoint,
56
- StaleCheckpointError,
57
62
  planIdentityOf,
58
63
  resolveHeadAt,
59
64
  resolveWorktreeRoot,
60
65
  sha256File,
66
+ StaleCheckpointError,
61
67
  type ExecutionApproval,
68
+ type WorkflowCheckpoint,
62
69
  } from "./workflow-state.ts";
63
70
  import { graphBlockForExecutor } from "./code-graph/prompts.ts";
64
- import {
65
- PANEL_WIDGET_KEY,
66
- derivePanelModel,
67
- deriveNextAction,
68
- deriveImplReviewLoopModel,
69
- formatImplReviewLoopSummaryLine,
70
- formatPanelSummaryLine,
71
- renderImplReviewLoopLines,
72
- renderPanelLines,
73
- themeImplReviewLoopLines,
74
- themePanelLines,
75
- } from "./panel.ts";
76
71
  import { resolveGraphMode } from "./code-graph/mode.ts";
77
72
  import {
78
- TERMINATION_QUESTION,
79
- TERMINATION_OPTIONS,
80
- TERMINATION_RECORDING_INSTRUCTIONS,
81
- defaultImplReviewers,
82
- implReviewerCountPromptLine,
83
- renderTerminationOptions,
84
- } from "./termination-prompt.ts";
85
- import {
86
- extractCoverage,
87
- latestPlanVersion,
88
73
  parseChecklist,
89
- parseImplItems,
90
- resolveImplStatuses,
91
- scanDoneMarkers,
92
- scanImplMarkers,
93
- scanCurrentIMarkers,
94
- resolveCurrentI,
95
- inferCurrentI,
96
- lintImplItems,
74
+ parsePlanTasks,
75
+ flattenTasks,
97
76
  type CheckItem,
98
- type ImplItem,
99
- type ImplMarkerState,
77
+ type PlanTasks,
100
78
  } from "./plan.ts";
79
+ import {
80
+ allTasksTerminal,
81
+ auditRollbackSet,
82
+ auditableChecks,
83
+ buildTaskView,
84
+ currentTask,
85
+ flattenTaskViews,
86
+ invalidateChecksForRolledBackTasks,
87
+ taskIsTerminal,
88
+ taskProgress,
89
+ taskProgressMap,
90
+ type TaskProgressMap,
91
+ type TaskView,
92
+ } from "./tasks.ts";
93
+ import { isAutoApproveEnabled as isAutoApproveEnabledLocal } from "./auto-approve.ts";
94
+ import {
95
+ DASHBOARD_WIDGET_KEY,
96
+ deriveDashboardModel,
97
+ formatDashboardSummaryLine,
98
+ formatElapsed,
99
+ renderDashboardLines,
100
+ renderDashboardTreeLines,
101
+ } from "./dashboard.ts";
102
+ import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditRoundResult } from "./auditor.ts";
103
+ import { staleReloadHint as probeStaleReload } from "./staleness.ts";
104
+ import { messaging } from "./messaging.ts";
105
+ import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
106
+ import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
107
+ import { resolveReviewerSpawn } from "./thinking-levels.ts";
108
+ import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
101
109
 
102
110
  export interface ExecState {
103
111
  planPath: string;
112
+ /** Verification checks (VC-###) — the audit's contract. */
104
113
  items: CheckItem[];
114
+ /** Parsed plan task model (kept for re-deriving the view). */
115
+ planTasks: PlanTasks;
116
+ /** Live task tree (the single source of progress). */
117
+ tasks: TaskView[];
118
+ /** True when the plan parsed through the legacy I-### fallback. */
119
+ legacyPlan: boolean;
105
120
  startedAt: string;
106
121
  usage: { inToks: number; outToks: number };
107
- implItems?: ImplItem[];
108
- implStatus?: Record<string, ImplMarkerState>;
109
- /** Plan-lint warning backing the panel's implWarning line. */
110
- implWarning?: string | null;
111
- /** Chrome language for panel/status strings (issue #3); undefined → "en". */
122
+ /** Chrome language for panel/status strings; undefined → "en". */
112
123
  uiLanguage?: UiLanguage;
113
- currentI?: string;
114
- goalWait?: GoalWaitState;
115
- /** v0.6.0: set while a delegated executor child owns the implementation.
116
- * Persisted subset only (modelSelector + startedAt); the AbortController is
117
- * runtime state kept in delegatedRuntime, never persisted. */
118
- delegate?: { modelSelector: string; startedAt: string };
124
+ /** Stall watchdog (v0.6.1): consecutive settled rounds without a task
125
+ * status change; auto-pause at the cap. */
126
+ stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
127
+ /** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
128
+ * only when a round outcome commits (never on discard/cancel) — and is the
129
+ * only piece persisted. */
130
+ audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
131
+ /** Execution-review loop (v0.8), memory-only: the attempt index names the
132
+ * per-round report files; consecutiveDiscards bounds the fingerprint
133
+ * re-run loop; inFlight owns the round's abort lifecycle. */
134
+ review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
135
+ /** Per-settle audit latch (v0.7.1): a settled round fires the completion
136
+ * audit at most once, so the turn_end / agent_before_settle / resume entry
137
+ * points cannot double-consume a round when several land in one settle.
138
+ * Created on demand by auditLatchOf(); every construction path may omit it. */
139
+ auditLatch?: { auditedThisSettle: boolean; activity: number };
140
+ }
141
+
142
+ /** One in-flight review round: owns its abort lifecycle, its fingerprint of
143
+ * the audited subject, and the per-round one-shot wake token (v0.8). */
144
+ interface InFlightReview {
145
+ controller: AbortController;
146
+ /** Budget round this attempt belongs to (audit.rounds + 1 at spawn). */
147
+ budgetRound: number;
148
+ /** Monotonic attempt ordinal; names the round report file. */
149
+ attempt: number;
150
+ fingerprint: string;
151
+ wakeSent: boolean;
119
152
  }
120
153
 
121
154
  /**
122
- * D-008 (issue #3): re-resolve the chrome language and repaint the panel and
123
- * status bar. Called by the plans tool right after a successful
124
- * `set-language` so an executing run switches language without a restart
125
- * (and without reading config on every render tick).
126
- *
127
- * Capability guard (implementation review F-001): partial contexts (some
128
- * command/test harnesses expose only notify/select/input) may lack
129
- * setStatus/theme — the refresh must stay a no-op there instead of throwing
130
- * into the caller's error path (updateStatusWidget assumes a full ui).
155
+ * D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
156
+ * and status bar right after a `set-language` change.
131
157
  */
132
158
  export function refreshUiLanguage(ctx: ExtensionContext): void {
133
159
  if (execution) execution.uiLanguage = resolveUiLanguage(ctx.cwd);
@@ -135,43 +161,35 @@ export function refreshUiLanguage(ctx: ExtensionContext): void {
135
161
  updateStatusWidget(ctx);
136
162
  }
137
163
 
138
- export interface GoalWaitState {
139
- noProgressRounds: number;
140
- waitRounds: number;
141
- /** Marker/progress snapshot of the last goal-wait round; null = baseline not set. */
142
- lastMarkers: string | null;
143
- paused: boolean;
144
- pausedReason?: string;
145
- }
146
-
147
- const GOAL_WAIT_MAX_NO_PROGRESS = 3;
148
- const GOAL_WAIT_MAX_WAITING = 6;
164
+ /** Consecutive no-progress rounds before the watchdog pauses (D-021). */
165
+ const STALL_MAX_ROUNDS = 3;
149
166
 
150
167
  let execution: ExecState | null = null;
151
168
 
152
- export const GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
169
+ export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
170
+ /** Legacy v0.6.0 continuation message type — filtered on restore. */
171
+ const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
153
172
 
154
- interface GoalWaitRuntime {
173
+ interface ContinuationRuntime {
155
174
  owner: ExecState;
156
175
  session: ExtensionContext["sessionManager"];
157
176
  handled: boolean;
158
177
  stopReason?: string;
159
- text: string;
160
178
  wakeId?: string;
161
179
  }
162
180
 
163
- // Dispatch identity belongs to a live session, never to a persisted checklist.
164
- let goalWaitRuntime: GoalWaitRuntime | null = null;
181
+ // Dispatch identity belongs to a live session, never to a persisted state.
182
+ let continuationRuntime: ContinuationRuntime | null = null;
165
183
 
166
- function resetGoalWaitRuntime(ctx: ExtensionContext): void {
167
- goalWaitRuntime = execution
168
- ? { owner: execution, session: ctx.sessionManager, handled: false, text: "" }
184
+ function resetContinuationRuntime(ctx: ExtensionContext): void {
185
+ continuationRuntime = execution
186
+ ? { owner: execution, session: ctx.sessionManager, handled: false }
169
187
  : null;
170
188
  }
171
189
 
172
- function currentGoalWaitRuntime(ctx: ExtensionContext): GoalWaitRuntime | null {
173
- return goalWaitRuntime?.owner === execution && goalWaitRuntime.session === ctx.sessionManager
174
- ? goalWaitRuntime
190
+ function currentContinuationRuntime(ctx: ExtensionContext): ContinuationRuntime | null {
191
+ return continuationRuntime?.owner === execution && continuationRuntime.session === ctx.sessionManager
192
+ ? continuationRuntime
175
193
  : null;
176
194
  }
177
195
 
@@ -185,17 +203,14 @@ export function consumePendingExecutionFlush(): boolean {
185
203
  return pending;
186
204
  }
187
205
 
188
- function requestExecutionFlush(_pi: ExtensionAPI, _ctx: ExtensionContext): void {
189
- // Unconditional defer. turn_end fires mid-run in a gap between agent
190
- // operations where isIdle() reads true; persistence happens only at the
191
- // drain points: agent_settled, the next before_agent_start, and stop/complete.
206
+ function requestExecutionFlush(): void {
192
207
  pendingExecutionFlush = true;
193
208
  }
194
209
 
195
- export function drainExecutionFlush(pi: ExtensionAPI, ctx: ExtensionContext): void {
210
+ export function drainExecutionFlush(ctx: ExtensionContext): void {
196
211
  if (!execution || !pendingExecutionFlush) return;
197
212
  pendingExecutionFlush = false;
198
- persist(pi);
213
+ persist(ctx);
199
214
  updateStatusWidget(ctx);
200
215
  }
201
216
 
@@ -209,19 +224,22 @@ export interface CheckpointExecutionLoad {
209
224
  doneVcIds?: string[];
210
225
  reverifyAll?: boolean;
211
226
  pausedReason?: string;
227
+ /** v0.6.1: true when the checkpoint parsed through the legacy fallback. */
228
+ legacyPlan?: boolean;
229
+ /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
230
+ * detected — resume requires a fresh handoff approval. */
231
+ legacyDelegate?: boolean;
212
232
  error?: string;
213
233
  }
214
234
 
215
235
  /**
216
- * Shared restore primitive (I-005/I-006): load the executing state from a run
217
- * checkpoint into THIS session. Authorization is kept only when the recorded
218
- * approval matches the current plan digest; a HEAD change keeps the
219
- * authorization but re-verifies previously verified VCs (D-011/F-001).
220
- * F-002: the loaded state is persisted to the current session IMMEDIATELY so
221
- * session_start/session_tree restore paths cannot silently clear it.
236
+ * Shared restore primitive: load the executing state from a run checkpoint
237
+ * into THIS session. Authorization is kept only when the recorded approval
238
+ * matches the current plan digest; a HEAD change keeps the authorization but
239
+ * re-opens previously closed tasks (D-023: reverifyAll → task statuses are
240
+ * dropped and re-run).
222
241
  */
223
242
  export function loadExecutionFromCheckpoint(
224
- pi: ExtensionAPI,
225
243
  ctx: ExtensionContext,
226
244
  runId: string,
227
245
  ): CheckpointExecutionLoad {
@@ -235,9 +253,6 @@ export function loadExecutionFromCheckpoint(
235
253
  return { status: "plan-missing", error: planPath ? `plan file vanished: ${planPath}` : "checkpoint has no plan identity" };
236
254
  }
237
255
  const planText = fs.readFileSync(planPath, "utf8");
238
- // F-002 (implementation review): the recorded plan identity is over BYTES —
239
- // an in-place edit at the same path must not inherit the authorization or
240
- // the verified VCs. Refuse the load and require a fresh handoff.
241
256
  if (sha256File(planPath) !== cp.plan.sha256) {
242
257
  return {
243
258
  status: "plan-mismatch" as const,
@@ -246,74 +261,107 @@ export function loadExecutionFromCheckpoint(
246
261
  }
247
262
  const items = parseChecklist(planText);
248
263
  if (items.length === 0) {
249
- return { status: "plan-missing", error: `${planPath} has no parsable verifier checklist` };
250
- }
251
- const implItems = parseImplItems(planText);
252
- const doneIds = new Set(cp.execution.doneVcIds);
253
- // D-011/F-001: an unchanged plan digest keeps the recorded authorization;
254
- // a changed HEAD under it forces re-verification of previously verified VCs.
255
- // F-006 (implementation review): an approval without a resolvable HEAD
256
- // recorded an unverifiable code state — re-verify instead of trusting.
264
+ return { status: "plan-missing", error: `${planPath} has no parsable verification checks` };
265
+ }
266
+ const planTasks = parsePlanTasks(planText);
267
+ if (planTasks.tasks.length === 0) {
268
+ return { status: "plan-missing", error: `${planPath} has no parsable tasks (## Tasks or legacy ## Implementation Items)` };
269
+ }
257
270
  const headNow = resolveHeadAt(ctx.cwd);
258
271
  const headUnverifiable = cp.execution.approval === null || cp.execution.approval.headAtApproval === null;
259
272
  const headChanged =
260
273
  cp.execution.approval !== null &&
261
274
  cp.execution.approval.headAtApproval !== null &&
262
275
  cp.execution.approval.headAtApproval !== headNow;
276
+ // D-023: reverifyAll re-opens every closed task (statuses dropped); a
277
+ // normal restore replays the persisted task progress map.
263
278
  const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
279
+ const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
280
+ const tasks = buildTaskView(planTasks, progress);
281
+ // v0.8: a review-cap pause SURVIVES the restore — the budget must stay
282
+ // bounded across restarts; only /plans-execute grants a fresh one. The
283
+ // legacy v0.7 prefix stays dual-matched for one release.
284
+ const wasReviewCapPause = isReviewCapPause(cp.execution.pausedReason);
285
+ // Any live round from the replaced session graph dies here (CF2-002).
286
+ abortInFlightReview();
264
287
  if (!reverifyAll) {
265
- for (const item of items) {
266
- if (doneIds.has(item.id)) item.done = true;
288
+ for (const id of cp.execution.doneVcIds) {
289
+ const item = items.find((candidate) => candidate.id === id);
290
+ if (item) item.done = true;
267
291
  }
268
292
  }
269
293
  execution = {
270
294
  planPath,
271
295
  items,
296
+ planTasks,
297
+ tasks,
298
+ legacyPlan: planTasks.legacy,
272
299
  startedAt: utcNow(),
273
300
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
274
- implItems,
275
- implStatus: { ...cp.execution.implStatus },
276
- implWarning: lintImplItems(planText),
277
301
  uiLanguage: resolveUiLanguage(ctx.cwd),
278
- currentI: cp.execution.currentI,
279
- goalWait: {
280
- noProgressRounds: 0,
281
- waitRounds: 0,
282
- lastMarkers: null,
283
- paused: cp.execution.pausedReason !== undefined,
284
- pausedReason: cp.execution.pausedReason,
302
+ // D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
303
+ // resume itself is the user's intent; the reason is surfaced in the
304
+ // resume brief instead. EXCEPT a review-cap pause (v0.8): it must
305
+ // survive restores paused and at its committed round count, or the
306
+ // 5-round budget would never bound anything across restarts.
307
+ stall: {
308
+ rounds: cp.execution.stallRounds ?? 0,
309
+ lastSnapshot: null,
310
+ paused: wasReviewCapPause,
311
+ pausedReason: wasReviewCapPause ? (cp.execution.pausedReason ?? undefined) : undefined,
285
312
  },
313
+ audit: {
314
+ rounds: cp.execution.audit?.rounds ?? 0,
315
+ failed: [],
316
+ undeterminable: cp.execution.audit?.undeterminable ?? [],
317
+ running: false,
318
+ },
319
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
320
+ auditLatch: { auditedThisSettle: false, activity: 0 },
286
321
  };
322
+ execution.stall.lastSnapshot = stallSnapshot();
287
323
  executionRunId = runId;
288
324
  bindRun(ctx.sessionManager, ctx.cwd, runId);
289
- resetGoalWaitRuntime(ctx);
290
- pendingExecutionFlush = false; // restored state: no inherited flush debt
325
+ resetContinuationRuntime(ctx);
326
+ pendingExecutionFlush = false;
291
327
  resetExecutionCompactionState(ctx);
292
- // v0.6.0 orphaned-delegate detection: a restart never carries a live child.
293
- // If the checkpoint/session carries delegate state, surface it so the user
294
- // knows the previous executor died mid-run (VC state is intact, resumable).
295
- if (cp.execution.delegate) {
328
+ // D-020: an orphaned v0.6.0 delegated executor never survives a restart.
329
+ // Its checkpoint delegate marker REFUSES the direct load — the run must
330
+ // re-enter through the execution handoff so the C-006 approval gate
331
+ // applies; execution restarts from the first task after re-approval.
332
+ const legacyDelegate = cp.execution.delegate !== undefined;
333
+ if (legacyDelegate) {
296
334
  try {
297
335
  ctx.ui.notify?.(
298
- `pi-plans: the previous delegated executor (${cp.execution.delegate.modelSelector}) did not finish before this session ended. Verified VC state is preserved; resume with /plans-execute.`,
336
+ "pi-plans: this run was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Re-approve via /plans-execute; execution restarts from the first task (the 0.6.0 progress record cannot map onto the task tree).",
299
337
  "warning",
300
338
  );
301
339
  } catch {
302
- /* notification is best-effort */
340
+ /* best-effort */
303
341
  }
342
+ // Refuse the load: no execution state may activate without the fresh
343
+ // C-006 handoff approval.
344
+ execution = null;
345
+ executionRunId = null;
346
+ return {
347
+ status: "no-execution",
348
+ legacyDelegate: true,
349
+ error: "orphaned v0.6.0 delegated executor; re-approve via /plans-execute (execution restarts from the first task)",
350
+ };
304
351
  }
305
352
  if (headChanged) {
306
353
  withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
307
354
  }
308
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
309
- persist(pi); // F-002: immediate session snapshot
355
+ persist(ctx);
310
356
  updateStatusWidget(ctx);
311
357
  return {
312
358
  status: "loaded",
313
359
  planPath,
314
- doneVcIds: [...doneIds],
360
+ doneVcIds: [...cp.execution.doneVcIds],
315
361
  reverifyAll,
316
362
  pausedReason: cp.execution.pausedReason,
363
+ legacyPlan: planTasks.legacy,
364
+ legacyDelegate,
317
365
  };
318
366
  }
319
367
 
@@ -327,7 +375,6 @@ interface ExecutionCompactionState {
327
375
  lastSuccessfulUsagePercent: number | null;
328
376
  lastSuccessfulAt: string | null;
329
377
  rearmPending: boolean;
330
- /** Terminal failure metadata is retained for diagnostics, not proactive retry. */
331
378
  terminalBackoffTokens: number | null;
332
379
  pendingStats: VccCompactionStats | null;
333
380
  pendingFollowUpPrompt: string | null;
@@ -387,63 +434,21 @@ export function handleExecutionTurnCompaction(ctx: ExtensionContext): void {
387
434
  consumeExecutionCompactionResumeGuard(ctx);
388
435
  }
389
436
 
390
- export function computeExecutionProgress(execution: ExecState): { done: number; total: number } {
391
- const implItems = execution.implItems ?? [];
392
- if (implItems.length) {
393
- const statuses = resolveImplStatuses(implItems, execution.items, execution.implStatus);
394
- const counted = implItems.filter((impl) =>
395
- execution.items.some((item) => extractCoverage(item.text).includes(impl.id)),
396
- );
397
- const total = counted.length > 0 ? counted.length : implItems.length;
398
- const vcDone = counted.filter((impl) => statuses[impl.id] === "vc-passed").length;
399
- const currentIndex = execution.currentI
400
- ? implItems.findIndex((impl) => impl.id === execution.currentI)
401
- : -1;
402
- return {
403
- done: Math.min(total, Math.max(vcDone, currentIndex < 0 ? 0 : currentIndex)),
404
- total,
405
- };
406
- }
407
- return {
408
- done: execution.items.filter((item) => item.done).length,
409
- total: execution.items.length,
410
- };
411
- }
412
-
413
- function formatElapsed(startedAt: string): string {
414
- const total = Math.max(0, Math.floor((Date.now() - Date.parse(startedAt)) / 1000));
415
- const h = String(Math.floor(total / 3600)).padStart(2, "0");
416
- const m = String(Math.floor((total % 3600) / 60)).padStart(2, "0");
417
- const sec = String(total % 60).padStart(2, "0");
418
- return `${h}:${m}:${sec}`;
419
- }
420
-
421
437
  function formatToks(tokens: number): string {
422
438
  const n = Math.max(0, Math.round(tokens));
423
439
  return n < 1000 ? String(n) : `${(n / 1000).toFixed(1)}k`;
424
440
  }
425
441
 
426
442
  export function formatExecutionStatusLine(execution: ExecState): string {
427
- const progress = computeExecutionProgress(execution);
428
- let line = `⌛ plans ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
429
- const goalWait = execution.goalWait;
430
- if (goalWait?.paused) {
431
- line += ` · ⏸ goal-wait paused (${goalWait.pausedReason ?? "paused"})`;
432
- } else if (goalWait && (goalWait.noProgressRounds > 0 || goalWait.waitRounds > 0)) {
433
- line += execChrome(execution.uiLanguage ?? "en").goalWait(goalWait.noProgressRounds, goalWait.waitRounds);
443
+ const progress = taskProgress(execution.tasks);
444
+ let line = `⌛ plans tasks ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
445
+ if (execution.stall.paused) {
446
+ line += ` · ⏸ paused (${execution.stall.pausedReason ?? "stalled"})`;
434
447
  }
435
448
  return line;
436
449
  }
437
450
 
438
- /** Approximation for "a subprocess is pending" (matches the exec loop's
439
- * `/waiting for/` backoff heuristic; F-008). Passed explicitly into the
440
- * shared model so the panel, status line and injection text agree. */
441
- export function executionIsWaiting(execution: ExecState): boolean {
442
- const gw = execution.goalWait;
443
- return gw !== undefined && !gw.paused && gw.waitRounds > 0;
444
- }
445
-
446
- /** Resolve the run topic for panel headers (falls back to the run id). */
451
+ /** Resolve the run topic for dashboard headers (falls back to the run id). */
447
452
  function panelTopic(ctx: ExtensionContext): string {
448
453
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
449
454
  if (active) {
@@ -454,155 +459,81 @@ function panelTopic(ctx: ExtensionContext): string {
454
459
  return "pi-plans";
455
460
  }
456
461
 
457
- /** Run info for the panel activity line (CQ1/D-005). Read by the in-flight
458
- * executionRunId so a stale/missing run record degrades to null (the model
459
- * then falls back to the bare phase word) instead of showing another run's
460
- * status. */
461
- function panelRunInfo(ctx: ExtensionContext): { status: string; created_at: string; updated_at: string } | null {
462
- if (!executionRunId) return null;
463
- // D-005 pointer-consistency: if the workdir's active pointer has moved to
464
- // another run (second session / external CLI mutation) while this
465
- // execution is live, the activity row degrades to the phase word rather
466
- // than mixing the new active run's topic (header) with the old run's
467
- // status (impl-review r1 F-001).
468
- if (resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id !== executionRunId) return null;
469
- const run = getRun(ctx.cwd, executionRunId);
470
- if (!run) return null;
471
- return { status: run.status, created_at: run.created_at, updated_at: run.updated_at };
472
- }
462
+ let dashboardRegistered = false;
463
+ let dashboardExpanded = false;
473
464
 
474
- let panelRegistered = false;
475
- let loopPanelRegistered = false;
465
+ /** Toggle the dashboard's expanded tree view (Ctrl+Shift+T). */
466
+ export function toggleDashboardExpanded(ctx: ExtensionContext): void {
467
+ dashboardExpanded = !dashboardExpanded;
468
+ dashboardRegistered = false; // force re-registration with the new mode
469
+ updateStatusWidget(ctx);
470
+ }
476
471
 
477
- /** Live implementation-review loop state for the panel: the widget stays
478
- * alive while the active run is done BUT its checkpoint is still in the
479
- * implementation-review phase (D-3). Read fresh on every call so post-write
480
- * redraws (index.ts turn-end updateStatusWidget) never show stale rounds
481
- * (D-8). Returns null once the phase flips to completed. */
482
- function implReviewLoopState(
483
- ctx: ExtensionContext,
484
- ): { topic: string; review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } } | null {
485
- // v0.6.0: resolveActiveRun only returns NON-TERMINAL runs, but this state
486
- // is by definition attached to a DONE run — fall back to the newest run of
487
- // any status (display-only) when the active resolution is null.
488
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd) ?? (() => {
489
- const latest = latestRun(ctx.cwd);
490
- return latest === null ? null : { run_id: latest.run_id, artifact_dir: latest.artifact_dir };
491
- })();
492
- if (!active) return null;
493
- if (getRun(ctx.cwd, active.run_id)?.status !== "done") return null;
494
- const load = loadCheckpoint(ctx.cwd, active.run_id);
495
- if (load.status !== "ok" || load.checkpoint.phase !== "implementation-review") return null;
496
- return { topic: panelTopic(ctx), review: load.checkpoint.implementationReview };
472
+ export function isDashboardExpanded(): boolean {
473
+ return dashboardExpanded;
497
474
  }
498
475
 
499
476
  /**
500
- * Register/update the fixed tasks' status panel (aboveEditor widget) for the
501
- * current execution, or unregister it when execution is gone. The panel and
502
- * the bottom status line share the same pure model (D-003/D-014/D-015). The
503
- * widget uses the factory form and reads the LIVE theme via `ui.theme` inside
504
- * render(width) — per-line width math happens on plain text first, then the
505
- * current theme is applied, so theme hot-swaps and resize never produce stale
506
- * colors or wrapped rows (F-004).
477
+ * Register/update the task dashboard (aboveEditor widget) for the current
478
+ * execution, or unregister it when execution is gone. The compact and the
479
+ * expanded tree view share one widget key so they never stack.
507
480
  */
508
481
  function updatePanelWidget(ctx: ExtensionContext): void {
509
- // Capability guard: older Pi hosts and test harness mocks may not expose
510
- // setWidget (the panel is a UI nicety, never a correctness dependency).
511
482
  if (!ctx.hasUI || typeof ctx.ui.setWidget !== "function") return;
512
- // Implementation-review loop widget (D-3): keeps the panel alive after
513
- // execution ends while the loop is live; unregisters when the phase
514
- // completes. Lives under the same widget key so the two widgets never
515
- // stack.
516
- const loop = execution === null ? implReviewLoopState(ctx) : null;
517
- if (execution === null && loop) {
518
- if (!loopPanelRegistered) {
519
- ctx.ui.setWidget(
520
- PANEL_WIDGET_KEY,
521
- (ui, _theme) => ({
522
- render(width: number) {
523
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
524
- // D-8 freshness: re-read the loop state per render; a phase flip to
525
- // completed renders an empty box until the next turn-end refresh
526
- // unregisters it (index.ts always calls updateStatusWidget then).
527
- const live = implReviewLoopState(ctx);
528
- if (!live) return [];
529
- const model = deriveImplReviewLoopModel(live.topic, live.review);
530
- const lines = renderImplReviewLoopLines(model, width);
531
- return theme ? themeImplReviewLoopLines(lines, theme as never) : lines;
532
- },
533
- }),
534
- { placement: "aboveEditor" },
535
- );
536
- loopPanelRegistered = true;
537
- }
538
- } else if (loopPanelRegistered) {
539
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
540
- loopPanelRegistered = false;
541
- }
542
483
  if (!execution) {
543
- if (panelRegistered) {
544
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
545
- panelRegistered = false;
484
+ if (dashboardRegistered) {
485
+ ctx.ui.setWidget(DASHBOARD_WIDGET_KEY, undefined);
486
+ dashboardRegistered = false;
546
487
  }
547
488
  return;
548
489
  }
549
- if (!panelRegistered) {
550
- ctx.ui.setWidget(
551
- PANEL_WIDGET_KEY,
552
- (ui, _theme) => ({
553
- render(width: number) {
554
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
555
- const current = execution;
556
- if (!current) return [];
557
- // F-004 (impl review r1): recompute the topic per render so a
558
- // cross-run restart without an intervening unregister cannot
559
- // show a stale box header.
560
- const model = derivePanelModel(current, panelTopic(ctx), executionIsWaiting(current), panelRunInfo(ctx));
561
- const lines = renderPanelLines(model, width);
562
- // Uniform-gray frame: │ borders never inherit the line color;
563
- // accents live between the borders only (themePanelLines).
564
- return theme ? themePanelLines(lines, model, theme) : lines;
565
- },
566
- }),
567
- { placement: "aboveEditor" },
568
- );
569
- panelRegistered = true;
570
- } else {
571
- // Factory components are re-created on every registration; content
572
- // updates flow through the closure reads at render time, so a no-op
573
- // re-set is unnecessary. Trigger one re-render via a cheap status
574
- // touch is NOT used — event-driven flush only (D-013).
575
- }
490
+ if (dashboardRegistered) return;
491
+ ctx.ui.setWidget(
492
+ DASHBOARD_WIDGET_KEY,
493
+ (ui, _theme) => ({
494
+ render(width: number) {
495
+ const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
496
+ const current = execution;
497
+ if (!current) return [];
498
+ const model = deriveDashboardModel(panelTopic(ctx), current.tasks, current.items, {
499
+ paused: current.stall.paused,
500
+ pausedReason: current.stall.pausedReason,
501
+ auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
502
+ auditFailed: current.audit.failed,
503
+ auditUndeterminable: current.audit.undeterminable,
504
+ reviewRunning: current.audit.running === true || current.review.inFlight !== null,
505
+ startedAt: current.startedAt,
506
+ usage: current.usage,
507
+ });
508
+ const lines = dashboardExpanded
509
+ ? renderDashboardTreeLines(model, width, theme)
510
+ : renderDashboardLines(model, width, theme);
511
+ return lines;
512
+ },
513
+ }),
514
+ { placement: "aboveEditor" },
515
+ );
516
+ dashboardRegistered = true;
576
517
  }
577
518
 
578
519
  export function updateStatusWidget(ctx: ExtensionContext): void {
579
520
  updatePanelWidget(ctx);
580
- if (execution) {
581
- // D-015: the status line derives from the same panel model.
582
- const line = formatPanelSummaryLine(
583
- derivePanelModel(execution, panelTopic(ctx), executionIsWaiting(execution), panelRunInfo(ctx)),
584
- );
585
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
521
+ if (execution && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
522
+ const model = deriveDashboardModel(panelTopic(ctx), execution.tasks, execution.items, {
523
+ paused: execution.stall.paused,
524
+ pausedReason: execution.stall.pausedReason,
525
+ auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
526
+ auditFailed: execution.audit.failed,
527
+ auditUndeterminable: execution.audit.undeterminable,
528
+ reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
529
+ });
530
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
586
531
  return;
587
532
  }
588
533
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
589
- if (active) {
590
- // Idle indicator depends on the run's lifecycle, not just its existence:
591
- // done reads as finished, abandoned as closed, stopped/accepted as paused.
592
- // v0.6.0: resolveActiveRun only returns NON-TERMINAL runs; for the pure
593
- // display line below, fall back to the newest run of any status so a
594
- // finished workdir still shows its last run's outcome.
534
+ if (active && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
595
535
  const status = getRun(ctx.cwd, active.run_id)?.status ?? latestRun(ctx.cwd)?.status;
596
536
  if (status === "done") {
597
- // D-3/D-015: while the implementation-review loop is live (checkpoint
598
- // still in the implementation-review phase), the status line mirrors
599
- // the loop box model instead of a bare "(done)".
600
- const loop = implReviewLoopState(ctx);
601
- if (loop) {
602
- const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
603
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
604
- return;
605
- }
606
537
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${active.run_id} (done)`));
607
538
  return;
608
539
  }
@@ -618,63 +549,67 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
618
549
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("warning", `⌛ plans: ${active.run_id}`));
619
550
  return;
620
551
  }
552
+ if (status === "executing") {
553
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
554
+ return;
555
+ }
556
+ if (status === "verifying") {
557
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `🔎 plans: ${active.run_id} (verifying)`));
558
+ return;
559
+ }
621
560
  if (status === "planning") {
622
- // Planning phase: 💬 while still in Q&A, 📝 once a PLAN draft exists
623
- // — kept until execution starts (then ⌛ takes over).
624
- const emoji = latestPlanVersion(active.artifact_dir) ? "📝" : "💬";
561
+ const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
625
562
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
626
563
  return;
627
564
  }
628
- // unknown status: no indicator.
629
- }
630
- // Terminal-only workdir: resolveActiveRun is null, but the display line
631
- // still reports the newest run's outcome (done/abandoned, plus its live
632
- // implementation-review loop).
633
- const terminalLatest = latestRun(ctx.cwd);
634
- if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
635
- if (terminalLatest.status === "done") {
636
- const loop = implReviewLoopState(ctx);
637
- if (loop) {
638
- const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
639
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
640
- return;
565
+ }
566
+ if (typeof ctx.ui?.setStatus === "function" && ctx.ui.theme) {
567
+ const terminalLatest = latestRun(ctx.cwd);
568
+ if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
569
+ if (terminalLatest.status === "done") {
570
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
571
+ } else {
572
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
641
573
  }
642
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
643
- } else {
644
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
574
+ return;
645
575
  }
646
- return;
576
+ ctx.ui.setStatus("pi-plans", undefined);
647
577
  }
648
- ctx.ui.setStatus("pi-plans", undefined);
649
578
  }
650
579
 
651
- function persist(pi: ExtensionAPI): void {
580
+ function parseLatestPlanExists(artifactDir: string): boolean {
581
+ try {
582
+ const names = fs.readdirSync(artifactDir);
583
+ return names.some((name) => /^PLAN_v\d+\.(md|markdown)$/i.test(name));
584
+ } catch {
585
+ return false;
586
+ }
587
+ }
588
+
589
+ function persist(ctx: ExtensionContext): void {
652
590
  if (!execution) return;
653
- pi.appendEntry("pi-plans-exec", {
591
+ messaging().appendEntry("pi-plans-exec", {
654
592
  planPath: execution.planPath,
655
593
  items: execution.items,
594
+ planTasks: execution.planTasks,
595
+ tasks: execution.tasks,
596
+ legacyPlan: execution.legacyPlan,
656
597
  startedAt: execution.startedAt,
657
598
  usage: execution.usage,
658
- implItems: execution.implItems,
659
- implStatus: execution.implStatus,
660
- implWarning: execution.implWarning ?? null,
661
- currentI: execution.currentI,
662
- goalWait: execution.goalWait,
663
- delegate: execution.delegate ?? null,
599
+ stall: execution.stall,
600
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
664
601
  });
665
602
  }
666
603
 
667
604
  /** Checkpoint bookkeeping for the executing run; best-effort for legacy runs
668
- * without checkpoints (their cross-session resume degrades to R-008 rules). */
669
- function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./workflow-state.ts").WorkflowCheckpoint) => import("./workflow-state.ts").WorkflowCheckpoint): void {
605
+ * without checkpoints. */
606
+ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: WorkflowCheckpoint) => WorkflowCheckpoint): void {
670
607
  if (!execution) return;
671
608
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
672
609
  if (!active || active.run_id !== executionRunId) return;
673
610
  try {
674
611
  mutateCheckpoint(ctx.cwd, active.run_id, mutator);
675
612
  } catch (error) {
676
- // F-005 (implementation review): ownership loss and revision staleness
677
- // must stop the advance, not vanish into the catch block.
678
613
  if (error instanceof OwnershipError || error instanceof StaleCheckpointError) throw error;
679
614
  /* legacy run or corrupt checkpoint: session snapshot still carries the loop */
680
615
  }
@@ -682,284 +617,109 @@ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./
682
617
 
683
618
  let executionRunId: string | null = null;
684
619
 
620
+ export interface StartExecutionInput {
621
+ planPath: string;
622
+ planTasks: PlanTasks;
623
+ items: CheckItem[];
624
+ }
625
+
685
626
  export async function startExecution(
686
- pi: ExtensionAPI,
687
627
  ctx: ExtensionContext,
688
- planPath: string,
689
- items: CheckItem[],
690
- implItems?: ImplItem[],
691
- opts?: StartExecutionOptions,
628
+ input: StartExecutionInput,
692
629
  ): Promise<void> {
630
+ // A fresh handoff replaces any live run — abort its in-flight review round first.
631
+ abortInFlightReview();
632
+ const tasks = buildTaskView(input.planTasks);
693
633
  execution = {
694
- planPath,
695
- items,
634
+ planPath: input.planPath,
635
+ items: input.items,
636
+ planTasks: input.planTasks,
637
+ tasks,
638
+ legacyPlan: input.planTasks.legacy,
696
639
  startedAt: utcNow(),
697
640
  usage: { inToks: 0, outToks: 0 },
698
- implItems: implItems ?? [],
699
- implStatus: {},
700
641
  uiLanguage: resolveUiLanguage(ctx.cwd),
701
- // F-001 (impl review r1): derive the plan-lint warning on the live
702
- // handoff path too, so the panel shows the ⚠ line immediately for a
703
- // zero-parse section instead of only after a checkpoint restore.
704
- implWarning: (() => {
705
- try {
706
- return lintImplItems(fs.readFileSync(planPath, "utf8"));
707
- } catch {
708
- return null;
709
- }
710
- })(),
711
- goalWait: { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
642
+ stall: { rounds: 0, lastSnapshot: null, paused: false },
643
+ audit: { rounds: 0, failed: [], undeterminable: [], running: false },
644
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
645
+ auditLatch: { auditedThisSettle: false, activity: 0 },
712
646
  };
713
- // Seed the marker baseline so the first quiet round is counted against a
714
- // real snapshot instead of counting unconditionally (F-006).
715
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
716
- resetGoalWaitRuntime(ctx);
717
- pendingExecutionFlush = false; // fresh run: no inherited flush debt
647
+ // Seed the watchdog baseline only after `execution` points at the new state
648
+ // (stallSnapshot reads the live execution).
649
+ execution.stall.lastSnapshot = stallSnapshot();
650
+ resetContinuationRuntime(ctx);
651
+ pendingExecutionFlush = false;
718
652
  resetExecutionCompactionState(ctx);
719
- persist(pi);
653
+ persist(ctx);
720
654
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
721
655
  executionRunId = active?.run_id ?? null;
722
656
  if (active) {
723
657
  bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
724
- // I-005: durable approval evidence — run + plan digest + HEAD at
725
- // approval (D-003/D-011). Sets phase executing via the state machine.
726
658
  try {
727
659
  const load = loadCheckpoint(ctx.cwd, active.run_id);
728
660
  if (load.status === "missing") {
729
661
  createCheckpoint(ctx.cwd, { runId: active.run_id, originWorkdir: ctx.cwd, workdir: ctx.cwd });
730
662
  }
731
663
  const approval: ExecutionApproval = {
732
- plan: planIdentityOf(path.resolve(planPath), 1),
664
+ plan: planIdentityOf(path.resolve(input.planPath), 1),
733
665
  worktree: resolveWorktreeRoot(ctx.cwd) ?? path.resolve(ctx.cwd),
734
666
  headAtApproval: resolveHeadAt(ctx.cwd),
735
667
  approvedAt: utcNow(),
736
668
  };
737
669
  mutateCheckpoint(ctx.cwd, active.run_id, (cp) => {
738
- // Plan refinement may not have recorded the plan identity yet.
739
670
  const withPlan = cp.plan === null ? { ...cp, plan: approval.plan } : cp;
740
- // The checkpoint may not carry accept-execute (legacy flow);
741
- // approval here came from the explicit handoff confirmation.
742
671
  const aligned = withPlan.nextAction === "accept-execute"
743
672
  ? withPlan
744
673
  : { ...withPlan, nextAction: "accept-execute" as const };
745
674
  return applyExecutionApproved(aligned, approval);
746
675
  });
747
- // Plan-lint entry point (execute handoff): a plan whose Implementation
748
- // Items section parses to zero items gets a durable run notice so the
749
- // execution panel's warning is backed by persisted evidence.
750
- lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(planPath));
676
+ lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(input.planPath));
751
677
  } catch (error) {
752
- // F-002 (implementation review): a plan-digest mismatch between the
753
- // recorded checkpoint plan and the approval must fail closed and
754
- // visibly — never silently execute without durable approval.
755
678
  if (error instanceof StateError && /does not match/.test(error.message)) throw error;
756
679
  /* legacy/corrupt checkpoint: run status still transitions below */
757
680
  }
758
681
  try {
759
682
  setRunStatus(ctx.cwd, active.run_id, "executing");
760
683
  } catch {
761
- /* status bookkeeping is best-effort */
684
+ /* best-effort */
762
685
  }
763
686
  }
764
- pi.sendMessage(
687
+ const progress = taskProgress(tasks);
688
+ messaging().sendMessage(
765
689
  {
766
690
  customType: "pi-plans-exec-start",
767
- content: `**pi-plans: executing** \`${planPath}\` — ${items.length} verifier item(s). Progress appears in the bottom status bar; mark verified items with \`[DONE:VC-xxx]\`.`,
691
+ content: `**pi-plans: executing** \`${input.planPath}\` — ${progress.total} task(s) in ${input.planTasks.legacy ? "legacy" : "task-tree"} mode, ${input.items.length} verification check(s). Report progress with the \`plans_update_task\` tool; the dashboard tracks every task (Ctrl+Shift+T expands the tree).`,
768
692
  display: true,
769
693
  },
770
694
  { triggerTurn: false },
771
695
  );
772
- // Delegated runtime (v0.6.0): one executor child implements the whole plan
773
- // while this tool call blocks; the parent mirrors VC progress from the
774
- // child's streamed assistant messages (R-9..R-12).
775
696
  updateStatusWidget(ctx);
776
- if (opts?.runtime && opts.runtime !== "current-session") {
777
- await runDelegatedExecution(pi, ctx, opts.runtime.modelSelector, opts.signal);
778
- }
779
- }
780
-
781
- /** Where a chosen execution runs: this session, or a delegated executor child. */
782
- export type ExecutionRuntime = "current-session" | { modelSelector: string };
783
-
784
- export interface StartExecutionOptions {
785
- /** "current-session" (default) or a delegated executor model selector. */
786
- runtime?: ExecutionRuntime;
787
- /** Tool-call abort signal, threaded into the delegated child. */
788
- signal?: AbortSignal;
789
- }
790
-
791
- /** Default delegated executor timeout when config omits executor_timeout_minutes. */
792
- const DELEGATE_DEFAULT_TIMEOUT_MINUTES = 60;
793
- void DELEGATE_DEFAULT_TIMEOUT_MINUTES;
794
-
795
- /** Live AbortController for the delegated executor child (runtime-only state). */
796
- let delegatedRuntime: AbortController | null = null;
797
-
798
- /** Abort the delegated executor child, if one is running (used by /plans-stop). */
799
- export function abortDelegatedExecutor(): boolean {
800
- if (delegatedRuntime === null) return false;
801
- delegatedRuntime.abort();
802
- return true;
803
- }
804
-
805
- function executorAgentPrompt(): string {
806
- try {
807
- const agentPath = new URL("../agents/executor.md", import.meta.url);
808
- return fs.readFileSync(agentPath, "utf8");
809
- } catch {
810
- return "You are a delegated plan executor in the pi-plans workflow. Implement the accepted plan autonomously and emit [DONE:VC-xxx] markers in your replies as verifier items pass.";
811
- }
812
697
  }
813
698
 
814
- function delegatedExecutorTimeoutMs(): number | undefined {
815
- try {
816
- const stateRoot = resolveStateRootOrNull(process.cwd());
817
- if (stateRoot === null) return undefined;
818
- const minutes = loadConfig(stateRoot).executor_timeout_minutes;
819
- if (typeof minutes !== "number" || minutes <= 0) return undefined;
820
- return minutes * 60 * 1000;
821
- } catch {
822
- return undefined;
823
- }
824
- }
825
-
826
- /**
827
- * Delegated execution (R-9..R-12): spawn ONE executor child for the whole
828
- * plan (write-capable tools, chosen model, PI_PLANS_EXECUTOR=1 + pinned run
829
- * id), stream its progress into the overlay, parse full-text assistant
830
- * messages for [DONE:VC-xxx]/[I-xxx] markers, and on exit verify the
831
- * remaining items. Abort/timeout → stopExecution (stopped, resumable);
832
- * clean exit with items left → stay executing (resumable under either
833
- * runtime); clean exit complete → normal completion flow.
834
- */
835
- async function runDelegatedExecution(
836
- pi: ExtensionAPI,
837
- ctx: ExtensionContext,
838
- modelSelector: string,
839
- parentSignal: AbortSignal | undefined,
840
- ): Promise<void> {
699
+ /** Persist the live task progress (called by the task status tool). */
700
+ export function persistTaskProgress(ctx: ExtensionContext): void {
841
701
  if (!execution) return;
842
- const controller = new AbortController();
843
- delegatedRuntime = controller;
844
- const relayAbort = () => controller.abort();
845
- if (parentSignal?.aborted) controller.abort();
846
- else parentSignal?.addEventListener("abort", relayAbort, { once: true });
847
- const lang = resolveUiLanguage(ctx.cwd);
848
- const overlay = ctx.mode === "tui" ? new RefineOverlayController("executor", [{ id: "executor" }], relayAbort, lang) : undefined;
849
- if (overlay) {
850
- try {
851
- overlay.open(refineOverlayContext(ctx), modelSelector);
852
- } catch {
853
- /* overlay is best-effort; the blocking call itself must not fail */
854
- }
855
- }
856
- execution.delegate = { modelSelector, startedAt: utcNow() };
857
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: execution?.delegate ?? null }));
858
- persist(pi);
859
- const runId = executionRunId;
860
- const remainingAtStart = execution.items.filter((item) => !item.done).map((item) => item.id);
861
- const task = [
862
- `Implement the accepted plan at ${execution.planPath} (workdir: ${ctx.cwd}).`,
863
- "Read the plan file first; it is the source of truth for scope, sequencing, and verification steps.",
864
- remainingAtStart.length > 0
865
- ? `Verifier items still open: ${remainingAtStart.join(", ")}. Emit [DONE:VC-xxx] markers in your replies as each item's stated evidence passes.`
866
- : "All verifier items already passed; verify the plan end-to-end and report.",
867
- execution.implItems?.length
868
- ? `Implementation items: ${execution.implItems.map((item) => item.id).join(", ")} — emit [I-###:implemented]/[I-###:validating] markers as you progress.`
869
- : "",
870
- "Finish with the structured summary your system prompt specifies.",
871
- ]
872
- .filter((line) => line !== "")
873
- .join("\n");
874
- let result: Awaited<ReturnType<typeof runPiSubagent>>;
875
- try {
876
- result = await runPiSubagent({
877
- systemPrompt: executorAgentPrompt(),
878
- task,
879
- cwd: ctx.cwd,
880
- model: modelSelector,
881
- tools: ["read", "write", "edit", "bash", "grep", "find", "ls"],
882
- envMarker: "executor",
883
- runId: runId ?? undefined,
884
- signal: controller.signal,
885
- timeoutMs: delegatedExecutorTimeoutMs(),
886
- onProgress: (event: SubagentProgressEvent) => {
887
- try {
888
- overlay?.update("executor", event);
889
- } catch {
890
- /* display must not fail the child runner */
891
- }
892
- if (
893
- event.type === "transcript"
894
- && event.phase === "end"
895
- && event.entryType === "assistant-text"
896
- && typeof event.text === "string"
897
- ) {
898
- mirrorDelegateMarkers(pi, ctx, event.text);
899
- }
702
+ // Any task-state change resets the stall watchdog baseline.
703
+ execution.stall.rounds = 0;
704
+ execution.stall.lastSnapshot = stallSnapshot();
705
+ withExecutionCheckpoint(ctx, (cp) =>
706
+ applyExecutionProgress(cp, {
707
+ tasks: taskProgressMap(execution!.tasks),
708
+ doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
709
+ stallRounds: execution!.stall.rounds,
710
+ audit: {
711
+ rounds: execution!.audit.rounds,
712
+ lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
713
+ undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
900
714
  },
901
- });
902
- } finally {
903
- delegatedRuntime = null;
904
- try {
905
- await overlay?.close();
906
- } catch {
907
- /* best-effort */
908
- }
909
- parentSignal?.removeEventListener("abort", relayAbort);
910
- }
911
- if (!result.ok) {
912
- const reason = result.timedOut
913
- ? `delegated executor timed out (${result.errorMessage ?? "no output"})`
914
- : result.cancelled || controller.signal.aborted
915
- ? "delegated executor aborted by user"
916
- : `delegated executor failed: ${result.errorMessage ?? "unknown error"}${result.stderr ? `; stderr: ${result.stderr.slice(0, 500)}` : ""}`;
917
- await stopExecution(pi, ctx, reason);
918
- return;
919
- }
920
- // Clean exit: land any markers from the final output text, then verify.
921
- mirrorDelegateMarkers(pi, ctx, result.output);
922
- if (isExecutionComplete()) {
923
- await completeExecution(pi, ctx);
924
- return;
925
- }
926
- // Items remain: keep the run executing (resumable via /plans-execute under
927
- // either runtime); the delegate bookkeeping is cleared so restarts do not
928
- // treat this as an orphaned child.
929
- if (execution) {
930
- execution.delegate = undefined;
931
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: null }));
932
- persist(pi);
933
- }
934
- const remaining = execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [];
935
- pi.sendMessage(
936
- {
937
- customType: "pi-plans-exec-delegate-exit",
938
- content: `**pi-plans: delegated executor exited with items remaining** — ${remaining.join(", ") || "(none)"}. Run stays executing; resume with /plans-execute (either runtime). Executor summary:\n${result.output.slice(0, 2000)}`,
939
- display: true,
940
- },
941
- { triggerTurn: false },
715
+ }),
942
716
  );
943
- ctx.ui.notify?.(`Delegated executor exited; ${remaining.length} verifier item(s) remain. Resume with /plans-execute.`, "warning");
944
- updateStatusWidget(ctx);
717
+ persist(ctx);
945
718
  }
946
719
 
947
- /** Apply VC/I markers parsed from a delegated child's full-text message. Exported for tests. */
948
- export function mirrorDelegateMarkers(pi: ExtensionAPI, ctx: ExtensionContext, text: string): void {
949
- const changedVc = applyDoneMarkers(text);
950
- const changedImpl = applyImplMarkers(text);
951
- applyCurrentIMarker(text);
952
- if (changedVc.length === 0 && changedImpl.length === 0) return;
953
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp));
954
- persist(pi);
955
- updateStatusWidget(ctx);
956
- }
957
-
958
- /** Record one assistant turn: accumulate usage and mark any completed items. */
720
+ /** Record one assistant turn: accumulate usage only (markers are gone). */
959
721
  export function recordExecutionTurn(
960
- pi: ExtensionAPI,
961
- _ctx: ExtensionContext,
962
- _completedIds: string[],
722
+ ctx: ExtensionContext,
963
723
  usage?: { input: number; output: number },
964
724
  ): void {
965
725
  if (!execution) return;
@@ -967,102 +727,645 @@ export function recordExecutionTurn(
967
727
  execution.usage.inToks += usage.input;
968
728
  execution.usage.outToks += usage.output;
969
729
  }
970
- // I-005: mirror progress into the run checkpoint so a different session
971
- // can resume with the verified VC/I set (R-004).
972
- withExecutionCheckpoint(_ctx, (cp) =>
730
+ withExecutionCheckpoint(ctx, (cp) =>
973
731
  applyExecutionProgress(cp, {
974
- doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
975
- implStatus: implStatusSnapshot(),
976
- currentI: execution!.currentI,
977
732
  usage: usage ? { inToks: usage.input, outToks: usage.output } : undefined,
978
733
  }),
979
734
  );
980
- requestExecutionFlush(pi, _ctx);
981
- updateStatusWidget(_ctx);
735
+ requestExecutionFlush();
736
+ updateStatusWidget(ctx);
982
737
  }
983
738
 
984
- function implStatusSnapshot(): Record<string, string> {
985
- const snapshot: Record<string, string> = {};
986
- if (!execution?.implItems) return snapshot;
987
- for (const item of execution.implItems) {
988
- const state = execution.implStatus?.[item.id];
989
- if (state) snapshot[item.id] = state;
990
- }
991
- return snapshot;
739
+ /** Test hook: replace the audit subagent with a deterministic function. */
740
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
741
+
742
+ export function __setAuditRunnerForTests(
743
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
744
+ ): void {
745
+ auditRunnerForTests = runner;
992
746
  }
993
747
 
994
748
  export function registerExecutionTurnHandlers(
995
- pi: ExtensionAPI,
749
+ ext: ExtensionAPI,
996
750
  onTurnEnd?: (ctx: ExtensionContext) => Promise<void> | void,
997
751
  ): void {
998
- // The turn_end projection does not carry usage; message_end delivers the
999
- // full assistant message, so cache it here and consume it per turn.
1000
752
  let lastAssistantUsage: { input: number; output: number } | null = null;
1001
- pi.on("agent_start", async (_event, ctx) => {
1002
- const runtime = currentGoalWaitRuntime(ctx);
753
+ ext.on("agent_start", async (_event, ctx) => {
754
+ const runtime = currentContinuationRuntime(ctx);
1003
755
  if (!runtime) return;
1004
756
  runtime.handled = false;
1005
757
  runtime.stopReason = undefined;
1006
- runtime.text = "";
758
+ // v0.7.1: a new agent run opens a new settle window (per-settle latch).
759
+ if (execution) resetSettleLatch();
1007
760
  });
1008
- pi.on("before_agent_start", async (_event, ctx) => {
1009
- const runtime = currentGoalWaitRuntime(ctx);
761
+ ext.on("before_agent_start", async (_event, ctx) => {
762
+ const runtime = currentContinuationRuntime(ctx);
1010
763
  if (runtime) runtime.wakeId = undefined;
1011
764
  });
1012
- pi.on("input", async (event, ctx) => {
1013
- if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(pi, ctx);
765
+ ext.on("input", async (event, ctx) => {
766
+ if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(ctx);
1014
767
  });
1015
- pi.on("agent_settled", async (_event, ctx) => {
1016
- drainExecutionFlush(pi, ctx);
1017
- maybeGoalWaitFollowUp(pi, ctx);
768
+ ext.on("agent_settled", async (_event, ctx) => {
769
+ drainExecutionFlush(ctx);
770
+ maybeContinuationFollowUp(ctx);
1018
771
  });
1019
- pi.on("session_shutdown", async (_event, ctx) => {
1020
- drainExecutionFlush(pi, ctx);
772
+ // v0.7.1: `agent_settled` is notification-only per pi's contract, so the
773
+ // audit fallback lives on `agent_before_settle` — the final ACTIONABLE
774
+ // boundary. It is what makes a terminal-but-unaudited run self-heal with
775
+ // zero user input, instead of stranding until a manual /plans-execute.
776
+ ext.on("agent_before_settle", async (_event, ctx) => {
777
+ if (!pendingAudit()) return;
778
+ const runtime = currentContinuationRuntime(ctx);
779
+ if (runtime?.handled) return;
780
+ // The continuation wake owns this settle: if the loop already woke the
781
+ // agent to fix rolled-back work, the audit waits for the next settle
782
+ // (Q-2 single-wake guarantee) rather than emitting a second wake.
783
+ if (runtime && !allTasksTerminal(runtime.owner.tasks)) {
784
+ maybeContinuationFollowUp(ctx);
785
+ return;
786
+ }
787
+ // Fully settled and still owed a review: launch the round under the
788
+ // mode rule — tui/rpc detach (the settle returns NOW and the overlay
789
+ // carries progress); print/json await inline so runtime teardown cannot
790
+ // kill the child. The round's own outcome routing drives the rest.
791
+ latchAuditThisSettle();
792
+ const chain = launchReviewRound(ctx);
793
+ if (chain) await chain;
794
+ });
795
+ ext.on("session_shutdown", async (_event, ctx) => {
796
+ drainExecutionFlush(ctx);
797
+ abortInFlightReview();
1021
798
  execution = null;
1022
799
  executionRunId = null;
1023
- goalWaitRuntime = null;
800
+ continuationRuntime = null;
1024
801
  lastAssistantUsage = null;
1025
802
  });
1026
- pi.on("message_end", async (event) => {
803
+ ext.on("message_end", async (event) => {
1027
804
  const message = event.message as { role?: string; usage?: { input?: number; output?: number } };
1028
805
  if (message?.role === "assistant" && message.usage) {
1029
806
  lastAssistantUsage = { input: message.usage.input ?? 0, output: message.usage.output ?? 0 };
1030
807
  }
1031
808
  });
809
+ // v0.7.1 (root cause B): the stall watchdog counted only task-status changes,
810
+ // so a round where the agent legitimately investigated (read code, gathered
811
+ // evidence) without closing a task looked identical to a dead agent. A
812
+ // SUCCESSFUL tool result is real progress; a failed/blocked tool is not, so
813
+ // an agent looping on the same error still trips the cap (F-005, Q-3).
814
+ ext.on("tool_result", async (event, _ctx) => {
815
+ if (!execution) return;
816
+ const result = event as { isError?: boolean; error?: unknown };
817
+ if (result.isError === true || result.error !== undefined) return;
818
+ auditLatchOf(execution).activity += 1; // Real progress: rebase the watchdog so this round counts as a change.
819
+ execution.stall.rounds = 0;
820
+ execution.stall.lastSnapshot = stallSnapshot();
821
+ });
1032
822
 
1033
- pi.on("turn_end", async (event, ctx) => {
1034
- const message = event.message as { role?: string; stopReason?: string; content?: Array<{ type: string; text?: string }> };
823
+ ext.on("turn_end", async (event, ctx) => {
824
+ const message = event.message as { role?: string; stopReason?: string };
1035
825
  if (!message || message.role !== "assistant") {
1036
826
  updateStatusWidget(ctx);
1037
827
  return;
1038
828
  }
1039
- const text = (message.content ?? [])
1040
- .filter((part) => part.type === "text")
1041
- .map((part) => part.text ?? "")
1042
- .join("\n");
1043
- const runtime = currentGoalWaitRuntime(ctx);
1044
- if (runtime) {
1045
- runtime.stopReason = message.stopReason;
1046
- runtime.text = text;
1047
- }
1048
- const changedIds = applyDoneMarkers(text);
1049
- const changedImpls = applyImplMarkers(text);
1050
- const changedCurrentI = applyCurrentIMarker(text);
829
+ const runtime = currentContinuationRuntime(ctx);
830
+ if (runtime) runtime.stopReason = message.stopReason;
1051
831
  const projection = (event.message as { usage?: { input?: number; output?: number } }).usage;
1052
832
  const raw = projection ?? lastAssistantUsage;
1053
- lastAssistantUsage = null; // consumed: never re-attribute a stale turn
833
+ lastAssistantUsage = null;
1054
834
  const usage = raw ? { input: raw.input ?? 0, output: raw.output ?? 0 } : undefined;
1055
- if (usage || changedIds.length > 0 || changedImpls.length > 0 || changedCurrentI) {
1056
- // Attribute this turn's usage now; `[DONE]` markers still only mark completion.
1057
- recordExecutionTurn(pi, ctx, changedIds, usage);
1058
- }
1059
- if (getExecution() && isExecutionComplete()) {
1060
- await completeExecution(pi, ctx);
835
+ if (usage) recordExecutionTurn(ctx, usage);
836
+ if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
837
+ latchAuditThisSettle();
838
+ const chain = launchReviewRound(ctx);
839
+ if (chain) await chain;
1061
840
  }
1062
841
  await onTurnEnd?.(ctx);
1063
842
  });
1064
843
  }
1065
844
 
845
+ /** ==== Execution-review loop (v0.8) ====
846
+ *
847
+ * When every task is terminal and checks are still owed, the run enters the
848
+ * `verifying` status and a DETACHED read-only reviewer round runs in the
849
+ * background (tui/rpc): the settle handler returns immediately and the
850
+ * executor is truly idle while the overlay shows live progress. In print/json
851
+ * modes the settle handler keeps AWAITING the round inline — runtime
852
+ * teardown at settle would otherwise kill a detached child and swallow the
853
+ * pause signal.
854
+ *
855
+ * Budget: `audit.rounds` counts COMMITTED rounds only (pass, fail, or
856
+ * undeterminable); discards and cancellations burn nothing. Undeterminable
857
+ * rounds self-schedule the retry inside the loop (no wake). Two consecutive
858
+ * fingerprint discards commit as an undeterminable round so the loop stays
859
+ * bounded. Exhaustion pauses in EVERY mode (fail-closed) with an in-band
860
+ * `pi-plans-review-paused` message; the ONLY fresh-budget surface is
861
+ * /plans-execute — ordinary input and session restores never refill.
862
+ *
863
+ * Lifecycle: each round owns a session-scoped AbortController (never
864
+ * ctx.signal, which is turn-scoped), aborted from session_shutdown,
865
+ * stopExecution, startExecution, and restoreFromSession. Outcomes are
866
+ * guarded by execution identity (`execution !== owner` → silent discard).
867
+ *
868
+ * Completion stays fail-closed AND never fail-open: a run completes only
869
+ * when every pending check was affirmatively passed. */
870
+ const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
871
+ /** v0.7 protocol value — dual-matched for one release so checkpoints written
872
+ * by older builds keep their cap pause recognized on restore. */
873
+ const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
874
+
875
+ function isReviewCapPause(reason: string | undefined | null): boolean {
876
+ if (!reason) return false;
877
+ return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
878
+ }
879
+
880
+ /** The currently-running (or self-scheduling) review chain; the sanctioned
881
+ * test seam awaits this. */
882
+ let activeReviewChain: Promise<void> | null = null;
883
+
884
+ function abortInFlightReview(): void {
885
+ const inFlight = execution?.review.inFlight;
886
+ if (inFlight) {
887
+ try {
888
+ inFlight.controller.abort();
889
+ } catch {
890
+ /* already aborted */
891
+ }
892
+ if (execution) execution.review.inFlight = null;
893
+ }
894
+ activeReviewChain = null;
895
+ }
896
+
897
+ function reviewOwed(ex: ExecState): boolean {
898
+ return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
899
+ }
900
+
901
+ function runDirOf(ctx: ExtensionContext): string | null {
902
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
903
+ return runId ? runDirPath(ctx.cwd, runId) : null;
904
+ }
905
+
906
+ function setRunStatusForReview(ctx: ExtensionContext, status: "verifying" | "executing"): void {
907
+ const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
908
+ if (!runId) return;
909
+ try {
910
+ const current = getRun(ctx.cwd, runId)?.status;
911
+ if (current !== status && current !== "done" && current !== "abandoned") {
912
+ setRunStatus(ctx.cwd, runId, status);
913
+ }
914
+ } catch {
915
+ /* best-effort */
916
+ }
917
+ }
918
+
919
+ /** Round fingerprint (Q-fingerprint-scope): plan digest + git HEAD +
920
+ * covered-file mtimes. A change between round start and resolve means the
921
+ * reviewer judged a subject that no longer exists — discard and re-run. */
922
+ function captureReviewFingerprint(ctx: ExtensionContext, ex: ExecState): string {
923
+ const parts: string[] = [];
924
+ try {
925
+ parts.push(createHash("sha256").update(fs.readFileSync(ex.planPath, "utf8")).digest("hex"));
926
+ } catch {
927
+ parts.push("plan-unreadable");
928
+ }
929
+ try {
930
+ parts.push(execSync("git rev-parse HEAD", { cwd: ctx.cwd, stdio: ["ignore", "pipe", "pipe"] }).toString().trim());
931
+ } catch {
932
+ parts.push("no-head");
933
+ }
934
+ const covered = new Set<string>();
935
+ for (const task of flattenTaskViews(ex.tasks)) {
936
+ for (const file of task.files ?? []) covered.add(file);
937
+ }
938
+ const mtimes: string[] = [];
939
+ for (const file of [...covered].sort()) {
940
+ try {
941
+ mtimes.push(`${file}:${fs.statSync(path.resolve(ctx.cwd, file)).mtimeMs}`);
942
+ } catch {
943
+ mtimes.push(`${file}:missing`);
944
+ }
945
+ }
946
+ parts.push(mtimes.join("|"));
947
+ return createHash("sha256").update(parts.join("\u0000")).digest("hex");
948
+ }
949
+
950
+ /** Round timeout: a committed round is minutes, never the 60-min subagent
951
+ * default — a hung child must surface as a spawn-failure round, not park the
952
+ * run in verifying for an hour (CF2-010). */
953
+ const REVIEW_ROUND_TIMEOUT_MS = 20 * 60 * 1000;
954
+
955
+ /** Reviewer-role pinning (Q-role-fallback): a CONFIRMED delegated role pins
956
+ * the spawn's model+thinking and labels the overlay with the role; an
957
+ * unconfirmed or current-session role inherits the session default with the
958
+ * "session default" label — a detached round NEVER opens the interactive
959
+ * first-use panel. */
960
+ function reviewSpawnProfile(): { model?: string; thinkingLevel?: string; label: string } {
961
+ try {
962
+ const reviewer = loadGlobalConfig().config.reviewer;
963
+ if (reviewer.mode !== "current-session" && reviewerReady(reviewer)) {
964
+ const spawn = resolveReviewerSpawn(reviewer);
965
+ if (spawn.modelSelector) {
966
+ return { model: spawn.modelSelector, thinkingLevel: spawn.thinkingLevel ?? undefined, label: spawn.label };
967
+ }
968
+ }
969
+ } catch {
970
+ /* fall through to the session default */
971
+ }
972
+ return { label: "session default" };
973
+ }
974
+
975
+ /** Engine-held lane state for the in-flight round: the reopen path builds a
976
+ * fresh one-shot controller seeded from THIS object, so the accumulated
977
+ * transcript survives ESC + reopen (CF2-001 / Q-reopen-seed). */
978
+ let reviewLane: RefineLaneState | null = null;
979
+ let reviewOverlay: RefineOverlayController | null = null;
980
+ let reviewModelLabel: string | undefined;
981
+
982
+ function freshReviewLane(attempt: number): RefineLaneState {
983
+ return {
984
+ id: `review-round-${attempt}`,
985
+ label: `Execution review round ${attempt}`,
986
+ status: "queued",
987
+ phase: "queued",
988
+ detail: "",
989
+ transcript: [],
990
+ currentTurnIndex: 0,
991
+ scrollOffset: 0,
992
+ followTranscript: true,
993
+ viewportHeight: 1,
994
+ };
995
+ }
996
+
997
+ /** Fresh controller per round (the controller is one-shot: closed latch,
998
+ * overlayPromise bail, terminal-lane early return — reuse drops progress).
999
+ * A UI failure must NEVER kill the round itself — best-effort only. */
1000
+ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
1001
+ if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
1002
+ try {
1003
+ const controller = new RefineOverlayController("auditor", [{ id: lane.id, label: lane.label }], () => {}, lang ?? "en");
1004
+ controller.seedLane(lane);
1005
+ controller.open(refineOverlayContext(ctx), modelLabel);
1006
+ return controller;
1007
+ } catch {
1008
+ return null;
1009
+ }
1010
+ }
1011
+
1012
+ /** The reopen surface (Task-3.4): rebuilds the overlay from engine-held lane
1013
+ * state; inert when no round is in flight. */
1014
+ export function reopenReviewOverlay(ctx: ExtensionContext): void {
1015
+ if (!execution?.review.inFlight || !reviewLane) return;
1016
+ const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
1017
+ if (controller) reviewOverlay = controller;
1018
+ }
1019
+
1020
+ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1021
+ const detail =
1022
+ ex.audit.failed.length > 0
1023
+ ? `failed: ${ex.audit.failed.join(", ")}`
1024
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1025
+ const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1026
+ pauseForStall(ctx, reason);
1027
+ // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
1028
+ // pauseForStall's ui.notify is optional and absent headless, so the pause
1029
+ // must also land in the session stream every mode can read.
1030
+ messaging().sendMessage(
1031
+ { customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
1032
+ { triggerTurn: false },
1033
+ );
1034
+ }
1035
+
1036
+ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1037
+ if (!execution) return;
1038
+ const ex = execution;
1039
+ // Skipped-pass checks resolve without a subagent round.
1040
+ for (const id of presolvedCheckIds(ex.items, ex.tasks)) {
1041
+ const item = ex.items.find((candidate) => candidate.id === id);
1042
+ if (item) item.done = true;
1043
+ }
1044
+ // Only auditable checks (with task coverage) gate completion; checks that
1045
+ // cover no task can never be verified and never block or complete.
1046
+ const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
1047
+ if (pendingChecks.length === 0) {
1048
+ await completeExecution(ctx);
1049
+ return;
1050
+ }
1051
+ if (ex.stall.paused || ex.review.inFlight) return;
1052
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1053
+ pauseReviewCap(ctx, ex);
1054
+ return;
1055
+ }
1056
+ // Phase transition: the executor is done with its tasks; the review loop
1057
+ // owns the run until it converges (or pauses at the cap).
1058
+ setRunStatusForReview(ctx, "verifying");
1059
+ const round: InFlightReview = {
1060
+ controller: new AbortController(),
1061
+ budgetRound: ex.audit.rounds + 1,
1062
+ attempt: ex.review.attempts + 1,
1063
+ fingerprint: captureReviewFingerprint(ctx, ex),
1064
+ wakeSent: false,
1065
+ };
1066
+ ex.review.attempts = round.attempt;
1067
+ ex.review.inFlight = round;
1068
+ ex.audit.running = true; // dashboard mirror (the v0.8 model lands with the overlay task)
1069
+ const spawn = reviewSpawnProfile();
1070
+ reviewModelLabel = spawn.label;
1071
+ const lane = freshReviewLane(round.attempt);
1072
+ reviewLane = lane;
1073
+ reviewOverlay = openReviewOverlay(ctx, lane, ex.uiLanguage, spawn.label);
1074
+ updateStatusWidget(ctx);
1075
+ let result: AuditRoundResult;
1076
+ try {
1077
+ result = auditRunnerForTests
1078
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
1079
+ : await runCompletionAudit(ctx, {
1080
+ planPath: ex.planPath,
1081
+ checklist: ex.items,
1082
+ tasks: ex.tasks,
1083
+ round: round.attempt,
1084
+ model: spawn.model,
1085
+ thinkingLevel: spawn.thinkingLevel,
1086
+ timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
1087
+ signal: round.controller.signal,
1088
+ onProgress: (event) => {
1089
+ // The engine owns the lane state; the live controller only repaints.
1090
+ applyRefineProgress(lane, event);
1091
+ reviewOverlay?.rerender();
1092
+ },
1093
+ });
1094
+ } catch (error) {
1095
+ if (ex.review.inFlight === round) {
1096
+ ex.review.inFlight = null;
1097
+ ex.audit.running = false;
1098
+ }
1099
+ void reviewOverlay?.close();
1100
+ reviewOverlay = null;
1101
+ messaging().sendMessage(
1102
+ { customType: "pi-plans-review-error", content: `pi-plans: execution review round threw: ${String(error)}`, display: true },
1103
+ { triggerTurn: false },
1104
+ );
1105
+ updateStatusWidget(ctx);
1106
+ return;
1107
+ }
1108
+ // Overlay terminal state + close (the controller is one-shot; the engine-held
1109
+ // lane keeps the transcript for a later reopen within this round).
1110
+ const cancelledResult = result !== null && typeof result === "object" && "cancelled" in result;
1111
+ try {
1112
+ applyRefineResult(lane, {
1113
+ ok: !cancelledResult,
1114
+ output: result && "report" in result ? result.report : "",
1115
+ stderr: "",
1116
+ turns: 0,
1117
+ ...(cancelledResult ? { cancelled: true as const } : {}),
1118
+ });
1119
+ } catch {
1120
+ /* cosmetic only */
1121
+ }
1122
+ void reviewOverlay?.close();
1123
+ reviewOverlay = null;
1124
+ await handleReviewOutcome(ctx, ex, round, result, pendingChecks.map((item) => item.id));
1125
+ }
1126
+
1127
+ async function handleReviewOutcome(
1128
+ ctx: ExtensionContext,
1129
+ owner: ExecState,
1130
+ round: InFlightReview,
1131
+ result: AuditRoundResult,
1132
+ pendingIds: string[],
1133
+ ): Promise<void> {
1134
+ // Identity guard: a restore, stop, or fresh handoff replaced the run —
1135
+ // drop this outcome silently (CF2-002).
1136
+ if (execution !== owner) return;
1137
+ if (owner.review.inFlight !== round) return;
1138
+ owner.review.inFlight = null;
1139
+ owner.audit.running = false;
1140
+ if (result !== null && typeof result === "object" && "cancelled" in result) {
1141
+ // Aborted by shutdown/stop/restore/tree-switch: no round, no budget, no wake.
1142
+ updateStatusWidget(ctx);
1143
+ return;
1144
+ }
1145
+ const outcome = result;
1146
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1147
+ const fingerprintNow = captureReviewFingerprint(ctx, owner);
1148
+ const coveredTaskIds = flattenTaskViews(owner.tasks).map((task) => task.id);
1149
+ if (fingerprintNow !== round.fingerprint) {
1150
+ // The audited subject moved under the reviewer (Q-C): discard, re-run
1151
+ // without budget burn; two consecutive discards commit as undeterminable
1152
+ // so a mutating user cannot loop the loop for free (Q-discard-bound).
1153
+ owner.review.consecutiveDiscards += 1;
1154
+ const runDir = runDirOf(ctx);
1155
+ if (runDir) {
1156
+ writeReviewRoundReport(runDir, {
1157
+ budgetRound: round.budgetRound,
1158
+ attempt: round.attempt,
1159
+ outcome: "discarded",
1160
+ passed: [],
1161
+ failed: [],
1162
+ undeterminable: pendingIds,
1163
+ discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
1164
+ fingerprintCaptured: round.fingerprint,
1165
+ fingerprintFound: fingerprintNow,
1166
+ coveredTaskIds,
1167
+ report: reportText,
1168
+ });
1169
+ }
1170
+ if (owner.review.consecutiveDiscards >= 2) {
1171
+ owner.review.consecutiveDiscards = 0;
1172
+ await commitReviewOutcome(
1173
+ ctx,
1174
+ owner,
1175
+ round,
1176
+ { passed: [], failed: [], undeterminable: pendingIds, report: `(two consecutive fingerprint discards — committed as an undeterminable round)\n\n${reportText}` },
1177
+ pendingIds,
1178
+ );
1179
+ return;
1180
+ }
1181
+ persist(ctx);
1182
+ updateStatusWidget(ctx);
1183
+ await maybeContinueReview(ctx, owner);
1184
+ return;
1185
+ }
1186
+ owner.review.consecutiveDiscards = 0;
1187
+ await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
1188
+ }
1189
+
1190
+ async function commitReviewOutcome(
1191
+ ctx: ExtensionContext,
1192
+ ex: ExecState,
1193
+ round: InFlightReview,
1194
+ outcome: { passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null,
1195
+ pendingIds: string[],
1196
+ ): Promise<void> {
1197
+ // The budget is charged only when an outcome commits — never on discard
1198
+ // or cancellation (CF2-003).
1199
+ ex.audit.rounds = round.budgetRound;
1200
+ const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
1201
+ const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
1202
+ // Anything the round neither passed nor failed is undeterminable: the
1203
+ // report omitted the check, spelled the verdict unreadably, or the
1204
+ // subagent never ran.
1205
+ const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
1206
+ const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
1207
+ const reportText = outcome?.report ?? "(review subagent failed to run)";
1208
+ {
1209
+ const runDir = runDirOf(ctx);
1210
+ if (runDir) {
1211
+ writeReviewRoundReport(runDir, {
1212
+ budgetRound: round.budgetRound,
1213
+ attempt: round.attempt,
1214
+ outcome: outcome === null
1215
+ ? "spawn-failed"
1216
+ : failed.length > 0
1217
+ ? "failed"
1218
+ : undeterminable.length > 0 ? "undeterminable" : "passed",
1219
+ passed,
1220
+ failed,
1221
+ undeterminable,
1222
+ fingerprintCaptured: round.fingerprint,
1223
+ coveredTaskIds,
1224
+ report: reportText,
1225
+ });
1226
+ }
1227
+ }
1228
+
1229
+ // Fail-closed completion: every pending check affirmatively passed. An
1230
+ // all-undeterminable round yields failed === [] — completing here would be
1231
+ // fail-open, marking a run done with nothing verified.
1232
+ if (passed.length === pendingIds.length) {
1233
+ ex.audit.failed = [];
1234
+ ex.audit.undeterminable = [];
1235
+ for (const id of passed) {
1236
+ const item = ex.items.find((candidate) => candidate.id === id);
1237
+ if (item) item.done = true;
1238
+ }
1239
+ withExecutionCheckpoint(ctx, (cp) =>
1240
+ applyExecutionProgress(cp, {
1241
+ tasks: taskProgressMap(ex.tasks),
1242
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1243
+ audit: { rounds: ex.audit.rounds, passed: true },
1244
+ }),
1245
+ );
1246
+ await completeExecution(ctx);
1247
+ return;
1248
+ }
1249
+ // Partial progress counts: a check affirmed this round is done even when a
1250
+ // sibling failed, so a later round only re-judges what is still open.
1251
+ for (const id of passed) {
1252
+ const item = ex.items.find((candidate) => candidate.id === id);
1253
+ if (item) item.done = true;
1254
+ }
1255
+ ex.audit.failed = failed;
1256
+ ex.audit.undeterminable = undeterminable;
1257
+ const rolledBack: string[] = [];
1258
+ for (const id of failed) {
1259
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1260
+ }
1261
+ // A rollback reopens work other checks were verifying; those checks must
1262
+ // stop claiming the run is satisfied there.
1263
+ if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1264
+ withExecutionCheckpoint(ctx, (cp) =>
1265
+ applyExecutionProgress(cp, {
1266
+ tasks: taskProgressMap(ex.tasks),
1267
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1268
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
1269
+ }),
1270
+ );
1271
+ if (rolledBack.length > 0) {
1272
+ // Rolling back is itself forward progress for the watchdog, but NOT for
1273
+ // audit.rounds: that counter stays monotonic so the repair loop is
1274
+ // bounded. The repair belongs to the executor — back to executing.
1275
+ ex.stall.rounds = 0;
1276
+ ex.stall.lastSnapshot = stallSnapshot();
1277
+ setRunStatusForReview(ctx, "executing");
1278
+ }
1279
+ persist(ctx);
1280
+ updateStatusWidget(ctx);
1281
+ if (failed.length > 0) {
1282
+ // v0.8 wake: per-round one-shot token — exactly one triggerTurn per
1283
+ // committed failed outcome, even when the round resolves long after the
1284
+ // settle that spawned it. The continuation runtime is deliberately
1285
+ // untouched: detached rounds outlive their settle.
1286
+ if (!round.wakeSent) {
1287
+ round.wakeSent = true;
1288
+ const openTasks = flattenTaskViews(ex.tasks)
1289
+ .filter((task) => !taskIsTerminal(task))
1290
+ .map((task) => task.id);
1291
+ const stranded = rolledBack.length === 0;
1292
+ const content = `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the review automatically.` : ""}`;
1293
+ messaging().sendMessage(
1294
+ {
1295
+ customType: "pi-plans-audit-failed",
1296
+ content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}\n\n---\n${reportText.slice(0, 4000)}`,
1297
+ display: true,
1298
+ },
1299
+ { triggerTurn: true },
1300
+ );
1301
+ }
1302
+ return; // The agent repairs; the next settle re-enters the loop.
1303
+ }
1304
+ // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1305
+ // wake, no message; the dashboard/overlay carries the round counter.
1306
+ await maybeContinueReview(ctx, ex);
1307
+ }
1308
+
1309
+ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promise<void> {
1310
+ if (execution !== ex) return;
1311
+ if (!reviewOwed(ex)) return;
1312
+ if (ex.stall.paused) return;
1313
+ if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
1314
+ pauseReviewCap(ctx, ex);
1315
+ return;
1316
+ }
1317
+ await startReviewRound(ctx);
1318
+ }
1319
+
1320
+ /** Launch a review round under the mode rule: detached (fire-and-forget with
1321
+ * the chain tracked for the test seam) in tui/rpc; awaited inline otherwise
1322
+ * (print/json settle must hold the runtime open through the round). Returns
1323
+ * the chain when the caller must await it, null when detached. */
1324
+ function launchReviewRound(ctx: ExtensionContext): Promise<void> | null {
1325
+ const detach = ctx.mode === "tui" || ctx.mode === "rpc";
1326
+ const chain = (async () => {
1327
+ await startReviewRound(ctx);
1328
+ })();
1329
+ activeReviewChain = chain;
1330
+ if (detach) {
1331
+ chain.catch(() => {
1332
+ /* surfaced via the review messages */
1333
+ });
1334
+ return null;
1335
+ }
1336
+ return chain;
1337
+ }
1338
+
1339
+ /** Deterministic seam: await the in-flight (or self-scheduling) review chain. */
1340
+ export function __awaitReviewRoundForTests(): Promise<void> {
1341
+ return activeReviewChain ?? Promise.resolve();
1342
+ }
1343
+
1344
+ /** When this module graph was first imported into the running pi process.
1345
+ * pi loads extensions once, so a fix written to disk mid-session stays
1346
+ * invisible until /reload — which is exactly why an unreadable audit verdict
1347
+ * deserves a /reload hint rather than a bare retry. */
1348
+ const extensionModuleLoadedAt = new Date();
1349
+
1350
+ /** The /reload advice when the extension on disk is newer than the copy this
1351
+ * process loaded, else null. Shared with /plans via src/staleness.ts so both
1352
+ * report the same answer from one probe. */
1353
+ function staleReloadHint(): string | null {
1354
+ try {
1355
+ const root = path.dirname(path.dirname(new URL(import.meta.url).pathname));
1356
+ return probeStaleReload(root, extensionModuleLoadedAt);
1357
+ } catch {
1358
+ return null;
1359
+ }
1360
+ }
1361
+
1362
+ /** True when the session can surface a pause to a human (D-022): interactive
1363
+ * TUI/RPC sessions that are not running under PI_PLANS_AUTO_APPROVE. */
1364
+ function isInteractiveSession(ctx: ExtensionContext): boolean {
1365
+ if ((ctx.mode !== "tui" && ctx.mode !== "rpc") || ctx.hasUI !== true) return false;
1366
+ return !isAutoApproveEnabledLocal();
1367
+ }
1368
+
1066
1369
  const EXECUTION_RESUME_CUSTOM_TYPE = "pi-plans-exec-resume";
1067
1370
 
1068
1371
  function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase): { settings: PiPlansVccSettings; runId: string; artifactDir: string } | null {
@@ -1073,18 +1376,19 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
1073
1376
  const run = getRun(ctx.cwd, active.run_id);
1074
1377
  if (!run) return null;
1075
1378
  if (phase === "planning" && run.status !== "planning") return null;
1076
- if (phase === "execution" && run.status !== "executing") return null;
1379
+ if (phase === "execution" && run.status !== "executing" && run.status !== "verifying") return null;
1077
1380
  scaffoldVccSettings(stateRoot);
1078
1381
  return { settings: loadVccSettings(stateRoot), runId: run.run_id, artifactDir: run.artifact_dir };
1079
1382
  }
1080
1383
 
1081
1384
  function executionVccContext(): PiPlansVccPhaseContext {
1385
+ const current = execution ? currentTask(execution.tasks) : null;
1082
1386
  return {
1083
1387
  phase: "execution",
1084
1388
  planPath: execution?.planPath ?? null,
1085
- currentI: execution?.currentI ?? null,
1389
+ currentI: current?.id ?? null,
1086
1390
  remainingVerifierIds: execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [],
1087
- implementationIds: execution?.implItems?.map((item) => item.id) ?? [],
1391
+ implementationIds: execution ? flattenTaskViews(execution.tasks).map((task) => task.id) : [],
1088
1392
  };
1089
1393
  }
1090
1394
 
@@ -1128,7 +1432,6 @@ export function buildExecutionCompactionResult(event: SessionBeforeCompactEvent,
1128
1432
  }
1129
1433
 
1130
1434
  export function handleExecutionBeforeCompact(
1131
- pi: ExtensionAPI,
1132
1435
  ctx: ExtensionContext,
1133
1436
  event: SessionBeforeCompactEvent,
1134
1437
  ): SessionBeforeCompactResult | undefined {
@@ -1159,7 +1462,7 @@ export function handleExecutionBeforeCompact(
1159
1462
  state.pendingStats = built.stats;
1160
1463
  state.pendingFollowUpPrompt = built.followUpPrompt;
1161
1464
  state.pendingContinueAfterThresholdCompact = built.settings.continueAfterThresholdCompact;
1162
- requestExecutionFlush(pi, ctx);
1465
+ requestExecutionFlush();
1163
1466
  return { compaction: built.compaction };
1164
1467
  }
1165
1468
 
@@ -1167,7 +1470,7 @@ function runtimePiVersion(ctx: ExtensionContext): unknown {
1167
1470
  return (ctx as ExtensionContext & { piVersion?: unknown }).piVersion ?? VERSION;
1168
1471
  }
1169
1472
 
1170
- export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1473
+ export async function handleExecutionCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1171
1474
  if (!execution) return;
1172
1475
  const state = ensureExecutionCompactionState(ctx);
1173
1476
  const stats = state.pendingStats;
@@ -1187,10 +1490,10 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
1187
1490
  if (!event.willRetry && stats) {
1188
1491
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
1189
1492
  if (followUpPrompt) {
1190
- await pi.sendUserMessage?.(followUpPrompt);
1493
+ await messaging().sendUserMessage(followUpPrompt);
1191
1494
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
1192
1495
  state.resumeGuard = true;
1193
- pi.sendMessage(
1496
+ messaging().sendMessage(
1194
1497
  {
1195
1498
  customType: EXECUTION_RESUME_CUSTOM_TYPE,
1196
1499
  content: EXECUTION_COMPACTION_RESUME_MESSAGE,
@@ -1200,18 +1503,15 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
1200
1503
  );
1201
1504
  }
1202
1505
  }
1203
- requestExecutionFlush(pi, ctx);
1506
+ requestExecutionFlush();
1204
1507
  updateStatusWidget(ctx);
1205
1508
  }
1206
1509
 
1207
-
1208
- export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1510
+ export function handleExecutionCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1209
1511
  if (!execution) return;
1210
1512
  const state = executionCompactionState(ctx);
1211
1513
  const terminal = isTerminalCompactionFailure(event);
1212
1514
  if (terminal) {
1213
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
1214
- // only after real growth or high-watermark pressure so the loop stops.
1215
1515
  if (state) {
1216
1516
  state.inFlight = false;
1217
1517
  state.resumeGuard = false;
@@ -1228,7 +1528,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
1228
1528
  ? "pi-plans: compaction found nothing to summarize; backing off until the session grows past the keep-recent window."
1229
1529
  : "pi-plans: compaction was aborted (provider interruption, user cancel, or a competing manual compact); backing off until the session grows or usage nears the window.";
1230
1530
  ctx.ui.notify(message, "info");
1231
- requestExecutionFlush(pi, ctx);
1531
+ requestExecutionFlush();
1232
1532
  return;
1233
1533
  }
1234
1534
  if (state) {
@@ -1245,7 +1545,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
1245
1545
  `pi-plans: compaction failed (${event.reason}); execution remains active and will wait for the next eligible turn.`,
1246
1546
  "warning",
1247
1547
  );
1248
- requestExecutionFlush(pi, ctx);
1548
+ requestExecutionFlush();
1249
1549
  }
1250
1550
 
1251
1551
  export function filterExecutionResumeMessages<T extends { customType?: string }>(messages: T[]): T[] {
@@ -1255,24 +1555,12 @@ export function filterExecutionResumeMessages<T extends { customType?: string }>
1255
1555
  // ---------------------------------------------------------------------------
1256
1556
  // Planning-phase compaction: Pi core owns scheduling; this hook customizes
1257
1557
  // active planning compact events with the same VCC builder used by execution.
1258
- // The two state machines are kept independent (different memory slot and
1259
- // snapshot key) so execution never bleeds into planning.
1260
1558
  // ---------------------------------------------------------------------------
1261
1559
 
1262
1560
  export const PLANNING_RUN_START_CUSTOM_TYPE = "pi-plans-run-start";
1263
1561
  export const PLANNING_PLAN_WRITTEN_CUSTOM_TYPE = "pi-plans-plan-written";
1264
1562
  const PLANNING_RESUME_CUSTOM_TYPE = "pi-plans-plan-resume";
1265
1563
 
1266
- // ---------------------------------------------------------------------------
1267
- // Pre-plan compaction: right after `plans start-run` creates a new planning
1268
- // run, the extension triggers one VCC compaction so the new plan starts on a
1269
- // lean context (LLM reasoning degrades with longer context; see PLAN
1270
- // preplan-compact). The pending flag is session-scoped and opportunistic: it
1271
- // is set by the start-run tool case and consumed by the plans tool_result
1272
- // hook in index.ts, which requests the extension-context compact action and
1273
- // resumes planning exactly once regardless of success or failure.
1274
- // ---------------------------------------------------------------------------
1275
-
1276
1564
  export { PLANNING_PREPLAN_COMPACT_HINT };
1277
1565
  export const PLANNING_PREPLAN_RESUME_CUSTOM_TYPE = "pi-plans-preplan-resume";
1278
1566
 
@@ -1292,11 +1580,8 @@ export function consumePrePlanCompactPending(ctx: ExtensionContext): PrePlanComp
1292
1580
  return pending;
1293
1581
  }
1294
1582
 
1295
- /** Hidden resume message after the pre-plan compaction settles (success or
1296
- * failure): Pi's manual compaction never continues the aborted turn, so the
1297
- * planning workflow is continued exactly once from here. */
1298
- export function sendPrePlanCompactResume(pi: ExtensionAPI): void {
1299
- pi.sendMessage?.(
1583
+ export function sendPrePlanCompactResume(ctx: ExtensionContext): void {
1584
+ messaging().sendMessage(
1300
1585
  {
1301
1586
  customType: PLANNING_PREPLAN_RESUME_CUSTOM_TYPE,
1302
1587
  content: "Continue planning.",
@@ -1313,7 +1598,6 @@ interface PlanningCompactionState {
1313
1598
  lastAttemptReason: "manual" | "threshold" | "overflow" | null;
1314
1599
  lastSuccessfulUsagePercent: number | null;
1315
1600
  lastSuccessfulAt: string | null;
1316
- /** Terminal "nothing to compact" backoff: tokens observed when Pi refused. */
1317
1601
  terminalBackoffTokens: number | null;
1318
1602
  pendingStats: VccCompactionStats | null;
1319
1603
  pendingFollowUpPrompt: string | null;
@@ -1341,8 +1625,6 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1341
1625
  if (message.includes("nothing to compact") || message.includes("already compacted") || message.includes("session too small")) {
1342
1626
  return { kind: "content" };
1343
1627
  }
1344
- // abort/stream class: explicit event names only, so that provider blips
1345
- // (network down, etc.) stay retryable.
1346
1628
  const abortPatterns = [
1347
1629
  "this operation was aborted",
1348
1630
  "aborted",
@@ -1354,20 +1636,12 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1354
1636
  if (abortPatterns.some((pattern) => message.includes(pattern))) {
1355
1637
  return { kind: "abort-stream" };
1356
1638
  }
1357
- // Aborted with no recognized message: still an abort-class terminal so the
1358
- // next eligible turn does not immediately retry the same operation.
1359
1639
  if (event.aborted === true) {
1360
1640
  return { kind: "abort-stream" };
1361
1641
  }
1362
1642
  return null;
1363
1643
  }
1364
1644
 
1365
- /** Session-scoped phase-local "compaction in flight" guard.
1366
- * - Set on `session_before_compact` for the phase attributed by the custom
1367
- * instructions hint; auto-compaction (no hint) marks both phases defensively.
1368
- * - Cleared on `session_compact` and `session_compact_failed`.
1369
- * - Retained so lifecycle events expose the same phase-local state to tests
1370
- * and future Pi core schema additions. */
1371
1645
  type CompactionPhase = "planning" | "execution";
1372
1646
 
1373
1647
  function compactionLifecycleStore(ctx: ExtensionContext): {
@@ -1396,18 +1670,11 @@ export function noteCompactionStarted(ctx: ExtensionContext, customInstructions:
1396
1670
  } else if (isExecutionCustomInstructions(customInstructions)) {
1397
1671
  store.execution = true;
1398
1672
  } else {
1399
- // Auto-compaction (threshold/overflow/manual without our hint) marks both.
1400
1673
  store.planning = true;
1401
1674
  store.execution = true;
1402
1675
  }
1403
1676
  }
1404
1677
 
1405
- /** Pi core's `SessionCompactEvent` / `SessionCompactFailedEvent` do not carry
1406
- * `customInstructions` in any emission site, so the END side has no way to
1407
- * know which phase the compaction belonged to. Clearing both phases is the
1408
- * safe default — the per-phase start side (above) already encodes the hint
1409
- * attribution. The hint parameter is retained for API symmetry and future
1410
- * Pi core schema additions. */
1411
1678
  export function noteCompactionEnded(ctx: ExtensionContext, _customInstructions: unknown): void {
1412
1679
  const store = compactionLifecycleStore(ctx);
1413
1680
  store.planning = false;
@@ -1431,15 +1698,11 @@ export function consumePlanningCompactionResumeGuard(ctx: ExtensionContext): boo
1431
1698
  }
1432
1699
 
1433
1700
  export function refreshPlanningCompactionCooldown(_ctx: ExtensionContext): void {
1434
- // Pi core owns scheduling; retained for lifecycle compatibility only.
1701
+ // Retained for lifecycle compatibility only.
1435
1702
  }
1436
1703
 
1437
1704
  export function requestPlanningCompaction(_ctx: ExtensionContext): void {
1438
- // Generic proactive pi-plans compaction is intentionally disabled. Manual,
1439
- // threshold, and overflow compactions are handled by session_before_compact.
1440
- // The single exception is the pre-plan compaction: index.ts requests the
1441
- // extension-context compact action from the plans tool_result hook right
1442
- // after start-run (see PLANNING_PREPLAN_COMPACT_HINT).
1705
+ // Manual, threshold, and overflow compactions are handled by session_before_compact.
1443
1706
  }
1444
1707
 
1445
1708
  function buildPlanningVccResult(event: SessionBeforeCompactEvent, ctx: ExtensionContext): VccCompactionBuildResult | null {
@@ -1464,7 +1727,6 @@ export function buildPlanningCompactionResult(event: SessionBeforeCompactEvent,
1464
1727
  }
1465
1728
 
1466
1729
  export function handlePlanningBeforeCompact(
1467
- pi: ExtensionAPI,
1468
1730
  ctx: ExtensionContext,
1469
1731
  event: SessionBeforeCompactEvent,
1470
1732
  ): SessionBeforeCompactResult | undefined {
@@ -1498,7 +1760,7 @@ export function handlePlanningBeforeCompact(
1498
1760
  return { compaction: built.compaction };
1499
1761
  }
1500
1762
 
1501
- export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1763
+ export async function handlePlanningCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1502
1764
  if (getExecution()) return;
1503
1765
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1504
1766
  const state = session.__planningCompaction;
@@ -1519,10 +1781,10 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1519
1781
  if (!event.willRetry && stats) {
1520
1782
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
1521
1783
  if (followUpPrompt) {
1522
- await pi.sendUserMessage?.(followUpPrompt);
1784
+ await messaging().sendUserMessage(followUpPrompt);
1523
1785
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
1524
1786
  state.resumeGuard = true;
1525
- pi.sendMessage(
1787
+ messaging().sendMessage(
1526
1788
  {
1527
1789
  customType: PLANNING_RESUME_CUSTOM_TYPE,
1528
1790
  content: "Continue planning.",
@@ -1534,16 +1796,13 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1534
1796
  }
1535
1797
  }
1536
1798
 
1537
-
1538
- export function handlePlanningCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1799
+ export function handlePlanningCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1539
1800
  if (getExecution()) return;
1540
1801
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1541
1802
  const state = session.__planningCompaction;
1542
1803
  if (!state) return;
1543
1804
  const terminal = isTerminalCompactionFailure(event);
1544
1805
  if (terminal) {
1545
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
1546
- // only after real growth or high-watermark pressure so the loop stops.
1547
1806
  state.inFlight = false;
1548
1807
  state.resumeGuard = false;
1549
1808
  state.cooldownActive = true;
@@ -1576,19 +1835,20 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
1576
1835
  return messages.filter((message) => message.customType !== PLANNING_RESUME_CUSTOM_TYPE && message.customType !== PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
1577
1836
  }
1578
1837
 
1579
- export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): Promise<void> {
1838
+ export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
1580
1839
  if (!execution) return;
1840
+ // A stopped run's in-flight review round dies with it (typed cancelled —
1841
+ // no budget, no wake).
1842
+ abortInFlightReview();
1581
1843
  resetExecutionCompactionState(ctx);
1582
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1583
1844
  pendingExecutionFlush = false;
1584
- persist(pi);
1585
- // Checkpoint first: withExecutionCheckpoint guards on the live execution.
1845
+ persist(ctx);
1586
1846
  withExecutionCheckpoint(ctx, (cp) => applyExecutionStopped(cp, reason));
1587
1847
  execution = null;
1588
1848
  executionRunId = null;
1589
- goalWaitRuntime = null;
1590
- pi.appendEntry("pi-plans-exec-cleared", { reason });
1591
- pi.sendMessage(
1849
+ continuationRuntime = null;
1850
+ messaging().appendEntry("pi-plans-exec-cleared", { reason });
1851
+ messaging().sendMessage(
1592
1852
  {
1593
1853
  customType: "pi-plans-exec-stop",
1594
1854
  content: `**pi-plans: execution stopped** — ${reason}`,
@@ -1607,84 +1867,76 @@ export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, rea
1607
1867
  updateStatusWidget(ctx);
1608
1868
  }
1609
1869
 
1610
- /** Apply [DONE:VC-xxx] markers from an assistant message. Returns changed ids. */
1611
- export function applyDoneMarkers(text: string): string[] {
1612
- if (!execution) return [];
1613
- const changed: string[] = [];
1614
- for (const id of scanDoneMarkers(text)) {
1615
- const item = execution.items.find((candidate) => candidate.id === id && !candidate.done);
1616
- if (item) {
1617
- item.done = true;
1618
- changed.push(id);
1619
- }
1620
- }
1621
- return changed;
1870
+ /** v0.7.1: the per-settle latch, created on demand so no execution-construction
1871
+ * path can leave it undefined (a missing latch must degrade to "no latch",
1872
+ * never throw inside a lifecycle handler). */
1873
+ function auditLatchOf(ex: ExecState): ExecState["auditLatch"] {
1874
+ if (!ex.auditLatch) ex.auditLatch = { auditedThisSettle: false, activity: 0 };
1875
+ return ex.auditLatch;
1876
+ }
1877
+
1878
+ function stallSnapshot(): string {
1879
+ if (!execution) return "";
1880
+ // v0.7.1: the snapshot carries the round's tool-activity counter, so a round
1881
+ // in which the agent legitimately did work (read code, run commands) counts
1882
+ // as progress even when no task changed status. Only a round with neither a
1883
+ // status change NOR a successful tool result is "no progress".
1884
+ return JSON.stringify({ tasks: taskProgressMap(execution.tasks), activity: auditLatchOf(execution).activity });
1622
1885
  }
1623
1886
 
1624
1887
  /**
1625
- * Apply [I-xxx:implemented|validating] markers from an assistant message.
1626
- * Unknown I-ids are silently ignored; later markers overwrite earlier ones.
1627
- * Returns the ids whose state actually changed.
1888
+ * v0.7.1: shared "the execution review is owed" predicate. Every entry point
1889
+ * that can start the audit (turn_end, agent_before_settle, restoreFromSession,
1890
+ * the resume path) goes through this so they can never disagree.
1891
+ *
1892
+ * Semantics (unchanged from restoreFromSession's guard, F-006): a check that
1893
+ * covers no task never gates completion, so only auditable checks count.
1628
1894
  */
1629
- export function applyImplMarkers(text: string): string[] {
1630
- if (!execution?.implItems?.length) return [];
1631
- const known = new Set(execution.implItems.map((impl) => impl.id));
1632
- execution.implStatus ??= {};
1633
- const changed: string[] = [];
1634
- for (const marker of scanImplMarkers(text)) {
1635
- if (!known.has(marker.id)) continue;
1636
- const previous = execution.implStatus[marker.id];
1637
- execution.implStatus[marker.id] = marker.state;
1638
- if (previous !== marker.state) changed.push(marker.id);
1639
- }
1640
- return changed;
1641
- }
1642
-
1643
- export function applyCurrentIMarker(text: string): boolean {
1644
- if (!execution?.implItems?.length) return false;
1645
- const markers = scanCurrentIMarkers(text);
1646
- const resolved = resolveCurrentI(execution.implItems, markers, execution.currentI);
1647
- if (!resolved || resolved === execution.currentI) return false;
1648
- execution.currentI = resolved;
1649
- return true;
1895
+ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1896
+ if (!ex) return false;
1897
+ // A paused run is never self-driven: the stall / review-cap pause is an
1898
+ // explicit "hand control back" signal, and resuming it is the user's call.
1899
+ // This also bounds the zero-input continue loop in agent_before_settle.
1900
+ if (ex.stall.paused) return false;
1901
+ if (ex.review.inFlight) return false;
1902
+ return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
1650
1903
  }
1651
1904
 
1652
- export function isExecutionComplete(): boolean {
1653
- return execution !== null && execution.items.length > 0 && execution.items.every((item) => item.done);
1905
+ /** v0.7.1: record that this settle already ran (or declined) its audit, so a
1906
+ * second entry point in the same settle cannot re-consume a round. */
1907
+ function latchAuditThisSettle(): void {
1908
+ if (execution) auditLatchOf(execution).auditedThisSettle = true;
1654
1909
  }
1655
1910
 
1656
- function goalWaitSnapshot(): string {
1657
- if (!execution) return "";
1658
- return JSON.stringify({
1659
- done: execution.items
1660
- .filter((item) => item.done)
1661
- .map((item) => item.id)
1662
- .sort()
1663
- .join("|"),
1664
- implStatus: execution.implStatus ?? {},
1665
- currentI: execution.currentI ?? null,
1666
- });
1911
+ /** v0.7.1: called on agent_start — a new agent run is a new settle window, so
1912
+ * the latch and the activity counter both reset here. */
1913
+ function resetSettleLatch(): void {
1914
+ if (!execution) return;
1915
+ const latch = auditLatchOf(execution);
1916
+ latch.auditedThisSettle = false;
1917
+ latch.activity = 0;
1667
1918
  }
1668
1919
 
1669
- function pauseGoalWait(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): void {
1920
+ function pauseForStall(ctx: ExtensionContext, reason: string): void {
1670
1921
  const ex = getExecution();
1671
- if (!ex?.goalWait) return;
1672
- ex.goalWait.paused = true;
1673
- ex.goalWait.pausedReason = reason;
1674
- persist(pi);
1922
+ if (!ex) return;
1923
+ ex.stall.paused = true;
1924
+ ex.stall.pausedReason = reason;
1925
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: reason }));
1926
+ persist(ctx);
1675
1927
  ctx.ui.notify?.(
1676
- `pi-plans: goal-wait paused (${reason}). Send any message or run /plans-execute to resume.`,
1928
+ `pi-plans: execution paused (${reason}). Send any message or run /plans-execute to resume.`,
1677
1929
  "warning",
1678
1930
  );
1679
1931
  updateStatusWidget(ctx);
1680
1932
  }
1681
1933
 
1682
- function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1934
+ function canWakeExecution(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1683
1935
  const compaction = executionCompactionState(ctx);
1684
- return currentGoalWaitRuntime(ctx) === runtime
1936
+ return currentContinuationRuntime(ctx) === runtime
1685
1937
  && (ctx.mode === "tui" || ctx.mode === "rpc")
1686
- && !isExecutionComplete()
1687
- && !runtime.owner.goalWait?.paused
1938
+ && !allTasksTerminal(runtime.owner.tasks)
1939
+ && !runtime.owner.stall.paused
1688
1940
  && ctx.isIdle()
1689
1941
  && !ctx.hasPendingMessages()
1690
1942
  && !ctx.signal?.aborted
@@ -1694,15 +1946,14 @@ function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): bool
1694
1946
  && compaction?.pendingFollowUpPrompt == null;
1695
1947
  }
1696
1948
 
1697
- function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1949
+ function sendContinuationWake(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1698
1950
  if (!canWakeExecution(ctx, runtime)) return false;
1699
1951
  try {
1700
- // Custom messages bypass before_agent_start, so carry fresh execution rules.
1701
1952
  const content = executionContextMessage(ctx);
1702
1953
  if (!content) return false;
1703
1954
  runtime.wakeId = randomUUID();
1704
- pi.sendMessage({
1705
- customType: GOAL_WAIT_CUSTOM_TYPE,
1955
+ messaging().sendMessage({
1956
+ customType: EXECUTION_CONTINUE_CUSTOM_TYPE,
1706
1957
  content,
1707
1958
  display: false,
1708
1959
  details: { wakeId: runtime.wakeId },
@@ -1710,120 +1961,161 @@ function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: Goal
1710
1961
  return true;
1711
1962
  } catch (error) {
1712
1963
  runtime.wakeId = undefined;
1713
- pauseGoalWait(pi, ctx, `continuation failed: ${String(error)}`);
1964
+ pauseForStall(ctx, `continuation failed: ${String(error)}`);
1714
1965
  return false;
1715
1966
  }
1716
1967
  }
1717
1968
 
1718
1969
  /** Only a fully settled agent run can need an extra wake, never a tool turn. */
1719
- function maybeGoalWaitFollowUp(pi: ExtensionAPI, ctx: ExtensionContext): void {
1720
- const runtime = currentGoalWaitRuntime(ctx);
1970
+ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
1971
+ const runtime = currentContinuationRuntime(ctx);
1721
1972
  if (!runtime || runtime.handled || !ctx.isIdle()) return;
1722
1973
  if (ctx.mode !== "tui" && ctx.mode !== "rpc") return;
1723
1974
  if (runtime.stopReason === "error" || runtime.stopReason === "aborted" || ctx.signal?.aborted) {
1724
1975
  runtime.handled = true;
1725
- pauseGoalWait(pi, ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1976
+ pauseForStall(ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1726
1977
  return;
1727
1978
  }
1979
+ // v0.7.1: a fully-terminal run that still owes an audit is NOT a
1980
+ // continuation case — the audit owns that state (agent_before_settle).
1981
+ // Return without waking: emitting EXECUTION_CONTINUE here would send the
1982
+ // agent back to redo work it has already finished (F-003).
1983
+ if (execution && allTasksTerminal(execution.tasks) && pendingAudit(execution)) return;
1728
1984
  if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
1729
1985
  runtime.handled = true;
1730
1986
  const ex = runtime.owner;
1731
- ex.goalWait ??= { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false };
1732
- const goalWait = ex.goalWait;
1733
- const snapshot = goalWaitSnapshot();
1734
- const changed = goalWait.lastMarkers !== null && snapshot !== goalWait.lastMarkers;
1735
- goalWait.lastMarkers = snapshot;
1987
+ const snapshot = stallSnapshot();
1988
+ const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
1989
+ ex.stall.lastSnapshot = snapshot;
1736
1990
  if (changed) {
1737
- goalWait.noProgressRounds = 0;
1738
- goalWait.waitRounds = 0;
1739
- } else if (/waiting for/i.test(runtime.text)) {
1740
- goalWait.waitRounds += 1;
1991
+ ex.stall.rounds = 0;
1741
1992
  } else {
1742
- goalWait.noProgressRounds += 1;
1743
- }
1744
- if (goalWait.noProgressRounds >= GOAL_WAIT_MAX_NO_PROGRESS) {
1745
- pauseGoalWait(pi, ctx, `no progress in ${goalWait.noProgressRounds} rounds`);
1746
- return;
1993
+ ex.stall.rounds += 1;
1747
1994
  }
1748
- if (goalWait.waitRounds >= GOAL_WAIT_MAX_WAITING) {
1749
- pauseGoalWait(pi, ctx, `waiting without progress for ${goalWait.waitRounds} rounds`);
1995
+ if (ex.stall.rounds >= STALL_MAX_ROUNDS) {
1996
+ pauseForStall(ctx, `no task-status change in ${ex.stall.rounds} rounds`);
1750
1997
  return;
1751
1998
  }
1752
- persist(pi);
1999
+ persist(ctx);
1753
2000
  updateStatusWidget(ctx);
1754
- // No await between the live gate and dispatch: another input cannot interleave.
1755
- sendGoalWaitWake(pi, ctx, runtime);
2001
+ sendContinuationWake(ctx, runtime);
1756
2002
  }
1757
2003
 
1758
2004
  export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
1759
- return messages.filter((message) => message.customType !== GOAL_WAIT_CUSTOM_TYPE
1760
- || (goalWaitRuntime?.owner === execution && goalWaitRuntime?.wakeId !== undefined
1761
- && (message.details as { wakeId?: unknown } | undefined)?.wakeId === goalWaitRuntime.wakeId));
2005
+ // v0.6.1: continuation wakes are one-shot; stale ones (including the
2006
+ // legacy v0.6.0 goal-wait type) never replay after a restart.
2007
+ return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
2008
+ && message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
2009
+ }
2010
+
2011
+ export function filterContinuationMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
2012
+ return filterGoalWaitMessages(messages);
1762
2013
  }
1763
2014
 
1764
- /** Called only for genuine user input or an explicit same-execution resume. */
1765
- export function resumeGoalWaitIfPaused(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
2015
+ /** Called for genuine user input or an explicit same-execution resume.
2016
+ * v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
2017
+ * refill the five-round budget (CF2-004); only /plans-execute
2018
+ * (resumeActiveExecution) is the explicit confirmation surface. Genuine
2019
+ * stall pauses still clear on input as before. */
2020
+ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
1766
2021
  const ex = getExecution();
1767
- if (!ex?.goalWait?.paused || !currentGoalWaitRuntime(ctx)) return false;
1768
- ex.goalWait.paused = false;
1769
- ex.goalWait.pausedReason = undefined;
1770
- ex.goalWait.noProgressRounds = 0;
1771
- ex.goalWait.waitRounds = 0;
1772
- ex.goalWait.lastMarkers = goalWaitSnapshot();
1773
- persist(pi);
2022
+ if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
2023
+ if (isReviewCapPause(ex.stall.pausedReason)) {
2024
+ // Surfaced once per input so the user is not left guessing why the run
2025
+ // stays paused; the pause itself and the budget survive untouched.
2026
+ ctx.ui.notify?.(
2027
+ "pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
2028
+ "warning",
2029
+ );
2030
+ return false;
2031
+ }
2032
+ ex.stall.paused = false;
2033
+ ex.stall.pausedReason = undefined;
2034
+ ex.stall.rounds = 0;
2035
+ ex.stall.lastSnapshot = stallSnapshot();
2036
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
2037
+ persist(ctx);
1774
2038
  updateStatusWidget(ctx);
1775
2039
  return true;
1776
2040
  }
1777
2041
 
1778
- export function resumeActiveExecution(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
1779
- if (!resumeGoalWaitIfPaused(pi, ctx)) return false;
1780
- const runtime = currentGoalWaitRuntime(ctx)!;
2042
+ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2043
+ // v0.8: /plans-execute is THE explicit confirmation surface for a
2044
+ // review-cap pause — the only place a fresh five-round budget is granted
2045
+ // (Q-confirm-surface). Ordinary input and session restores never refill.
2046
+ const pausedEx = getExecution();
2047
+ if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
2048
+ pausedEx.stall.paused = false;
2049
+ pausedEx.stall.pausedReason = undefined;
2050
+ pausedEx.stall.rounds = 0;
2051
+ pausedEx.stall.lastSnapshot = stallSnapshot();
2052
+ pausedEx.audit.rounds = 0;
2053
+ pausedEx.audit.failed = [];
2054
+ pausedEx.audit.undeterminable = [];
2055
+ withExecutionCheckpoint(ctx, (cp) =>
2056
+ applyExecutionProgress(cp, {
2057
+ tasks: taskProgressMap(pausedEx.tasks),
2058
+ audit: { rounds: 0, lastResult: undefined },
2059
+ pausedReason: null,
2060
+ }),
2061
+ );
2062
+ persist(ctx);
2063
+ updateStatusWidget(ctx);
2064
+ messaging().sendMessage(
2065
+ {
2066
+ customType: "pi-plans-review-budget-granted",
2067
+ content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
2068
+ display: true,
2069
+ },
2070
+ { triggerTurn: false },
2071
+ );
2072
+ const grantChain = launchReviewRound(ctx);
2073
+ if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
2074
+ return true;
2075
+ }
2076
+ // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
2077
+ // to `return false` here, so `/plans-execute` answered "already executing"
2078
+ // and the run stayed stranded until a full re-entry or a session restore.
2079
+ // It is not paused, so the pause path below cannot see it — check it first
2080
+ // and run the owed review instead of reporting "nothing to resume".
2081
+ if (!execution?.stall.paused && pendingAudit()) {
2082
+ latchAuditThisSettle();
2083
+ const chain = launchReviewRound(ctx);
2084
+ if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
2085
+ return true;
2086
+ }
2087
+ if (!resumeGoalWaitIfPaused(ctx)) return false;
2088
+ const runtime = currentContinuationRuntime(ctx)!;
1781
2089
  if (canWakeExecution(ctx, runtime)) {
1782
2090
  runtime.handled = true;
1783
- sendGoalWaitWake(pi, ctx, runtime);
2091
+ sendContinuationWake(ctx, runtime);
1784
2092
  }
1785
2093
  return true;
1786
2094
  }
1787
2095
 
1788
- export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext): Promise<void> {
2096
+ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1789
2097
  if (!execution) return;
1790
2098
  resetExecutionCompactionState(ctx);
1791
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1792
2099
  pendingExecutionFlush = false;
1793
- persist(pi);
1794
-
1795
- const summary = execution.items.map((item) => `- ✅ \`${item.id}\` ${item.text.split(";")[0]}`).join("\n");
2100
+ persist(ctx);
2101
+ const flat = flattenTaskViews(execution.tasks);
2102
+ const summary = flat
2103
+ .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
2104
+ .join("\n");
1796
2105
  const planPath = execution.planPath;
1797
- // Checkpoint first (live-execution guard), then clear the session state.
1798
2106
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
1799
2107
  execution = null;
1800
2108
  executionRunId = null;
1801
- goalWaitRuntime = null;
1802
- pi.appendEntry("pi-plans-exec-cleared", { reason: "complete" });
1803
- // Post-execution goal-running continuation: in interactive sessions, attach
1804
- // the continuation block and trigger a new turn so the agent immediately
1805
- // enters the implementation-review loop. Headless sessions keep the silent
1806
- // completion behavior. Both completeExecution call sites (turn_end and the
1807
- // restoreFromSession recovery path) share this behavior.
1808
- const interactive = ctx.hasUI === true;
1809
- // Skill-aware continuation: the reviewer-count default follows the active
1810
- // run's skill (D-1/D-4), so the prompt names the run's own recommended
1811
- // count instead of a static guess.
1812
- const activeRunForPrompt = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1813
- const content = interactive
1814
- ? `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}\n\n${ameliorationPromptText(activeRunForPrompt?.skill)}`
1815
- : `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}`;
1816
- pi.sendMessage(
2109
+ continuationRuntime = null;
2110
+ messaging().appendEntry("pi-plans-exec-cleared", { reason: "complete" });
2111
+ messaging().sendMessage(
1817
2112
  {
1818
2113
  customType: "pi-plans-complete",
1819
- content,
2114
+ content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}`,
1820
2115
  display: true,
1821
2116
  },
1822
- { triggerTurn: interactive },
2117
+ { triggerTurn: false },
1823
2118
  );
1824
- if (interactive) {
1825
- pi.appendEntry("pi-plans-ameliorate", { planPath, phase: "goal-started", rounds: null, currentRound: 0 });
1826
- }
1827
2119
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1828
2120
  if (active) {
1829
2121
  try {
@@ -1835,55 +2127,45 @@ export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext)
1835
2127
  updateStatusWidget(ctx);
1836
2128
  }
1837
2129
 
1838
- /** Instructions appended to the post-execution completion message in
1839
- * interactive sessions, telling the agent to enter the goal-running
1840
- * implementation-review loop. Skill-aware: the reviewer-count question's
1841
- * recommended option follows the run's skill (plan-big / plan-with-refs → 3,
1842
- * others → 1). Termination options are single-sourced from
1843
- * src/termination-prompt.ts (shared with the ask_choice trailing branch). */
1844
- export function ameliorationPromptText(skill: string | undefined): string {
1845
- return `---
1846
- Goal-running continuation: immediately ask the user now via ask_choice (autoComplete: false, in the session language) the termination question: "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(skill)} Then keep running the implementation-review loop without asking whether to continue; the goal-wait option keeps the loop running until no unpassed VCs remain.`;
1847
- }
1848
-
1849
2130
  /** Injection text for before_agent_start while executing. */
1850
2131
  export function executionContextMessage(ctx: ExtensionContext): string | null {
1851
2132
  if (!execution) return null;
1852
- const remaining = execution.items.filter((item) => !item.done);
1853
- const list =
1854
- remaining.map((item) => `- \`${item.id}\` ${item.text}`).join("\n") || "(none — report completion now)";
1855
- // Live read: the injected guidance and the tool wrappers share the same
1856
- // tri-state, so they can never contradict each other mid-run.
2133
+ const flat = flattenTaskViews(execution.tasks);
2134
+ const open = flat.filter((task) => !taskIsTerminal(task));
2135
+ const cur = currentTask(execution.tasks);
2136
+ const currentWave = cur?.wave ?? 1;
2137
+ const inWave = open.filter((task) => task.wave === currentWave);
2138
+ const waveList = inWave.map((task) => `- ${task.id}${task.children.length ? ` (${task.children.map((c) => c.id).join(", ")})` : ""}: ${task.title}${task.files.length ? ` — files: ${task.files.join(", ")}` : ""}`).join("\n") || "(none — take the next wave)";
2139
+ const progress = taskProgress(execution.tasks);
2140
+ const vcDone = execution.items.filter((item) => item.done).length;
1857
2141
  const mode = resolveGraphMode(ctx?.cwd ?? process.cwd());
1858
2142
  const graphLine =
1859
2143
  mode === "config-unavailable"
1860
- ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi_plans/config.json is repaired]`
2144
+ ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
1861
2145
  : graphBlockForExecutor(mode === "enabled");
1862
- // F-002 (impl review r1): the next-action line is hoisted out of the
1863
- // implItems ternary so plans without implementation items get the same
1864
- // same-source guidance the panel shows.
1865
- const nextActionLine = `\nSuggested next action (displayed in the pi-plans panel): ${deriveNextAction(execution, executionIsWaiting(execution), resolveImplStatuses(execution.implItems ?? [], execution.items, execution.implStatus), remaining, execution.currentI)}`;
1866
- const implementationItems = execution.implItems?.length
1867
- ? `\nImplementation items: ${execution.implItems.map((item) => item.id).join(", ")}${execution.currentI ? `\nCurrent implementation item: \`${execution.currentI}\`` : ""}\nWhen beginning an implementation item, emit its current anchor exactly once as \`[I-###:current]\`; then use \`[I-###:implemented]\` or \`[I-###:validating]\` for progress.`
2146
+ const rollbackNote = execution.audit.failed.length > 0
2147
+ ? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
1868
2148
  : "";
1869
2149
  return `[PI-PLANS EXECUTION — write access enabled]
1870
- Implement the accepted plan at ${execution.planPath} (${execution.items.length - remaining.length}/${execution.items.length} verifier items done).
2150
+ Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
2151
+
2152
+ Current wave ${currentWave} open tasks:
2153
+ ${waveList}
1871
2154
 
1872
- Remaining verifier items:
1873
- ${list}${implementationItems}${nextActionLine}
2155
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
1874
2156
 
1875
2157
  ${graphLine}
1876
2158
 
1877
2159
  Execution rules:
1878
- - Implement implementation items in dependency order; grow the change in layers — smallest end-to-end slice first, then stack each new capability on top of what already works.
1879
- - Report implementation-item progress with lightweight markers in your reply: write \`[I-001:implemented]\` when an item's code is done, \`[I-001:validating]\` when you start verifying it. The execution status bar tracks these states.
1880
- - For subprocess-backed verification, when a step starts a subprocess and needs its result before verifying, use literal \`waiting for\` with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s; restart at 5s for each new subprocess.
1881
- - Simplest implementation that fully meets the item: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1882
- - Architectural decisions are for the long term: no stopgaps. Do not add backward-compatibility layers, fallbacks, or migrations — remove the obsolete paths this change obsoletes.
1883
- - Prefer established, well-maintained libraries when they reduce complexity or improve reliability; before writing your own implementation or adding a package, check the project's existing dependencies (docs and types) — never reimplement common functionality without a clear reason.
1884
- - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists; when unsure, skip and emit \`[test skipped: <name>, add when <trigger>]\`.
1885
- - After verifying an item's pass condition with its stated evidence, include \`[DONE:VC-xxx]\` in your reply.
1886
- - When every item is done, report a completion summary.`;
2160
+ - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
2161
+ - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
2162
+ - Close subtasks before their parent; a parent is auditable only when every child is terminal.
2163
+ - When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
2164
+ - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
2165
+ - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
2166
+ - Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
2167
+ - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists.
2168
+ - For subprocess-backed verification, when a step needs a subprocess result before proceeding, poll with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s.`;
1887
2169
  }
1888
2170
 
1889
2171
  interface SessionEntry {
@@ -1895,24 +2177,21 @@ interface SessionEntry {
1895
2177
 
1896
2178
  /**
1897
2179
  * Rebuild execution state from the session on start/resume. Finds the last
1898
- * pi-plans-exec snapshot, then re-scans assistant messages after it for
1899
- * [DONE:VC-xxx] markers so progress survives restarts.
2180
+ * pi-plans-exec snapshot; the task tree is rebuilt from the persisted
2181
+ * snapshot (tool-driven progress survives restarts without text replay).
1900
2182
  */
1901
- export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
1902
- pendingExecutionFlush = false; // no flush debt survives a restart
1903
- goalWaitRuntime = null;
2183
+ export async function restoreFromSession(ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
2184
+ pendingExecutionFlush = false;
2185
+ continuationRuntime = null;
1904
2186
  resetExecutionCompactionState(ctx);
1905
- let snapshotIndex = -1;
1906
2187
  let snapshot: ExecState | null = null;
1907
2188
  for (let i = entries.length - 1; i >= 0; i--) {
1908
2189
  const entry = entries[i];
1909
2190
  if (entry.type === "custom" && entry.customType === "pi-plans-exec" && entry.data) {
1910
2191
  snapshot = entry.data;
1911
- snapshotIndex = i;
1912
2192
  break;
1913
2193
  }
1914
2194
  if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1915
- // Execution was explicitly stopped or completed after the last snapshot.
1916
2195
  execution = null;
1917
2196
  updateStatusWidget(ctx);
1918
2197
  return;
@@ -1923,78 +2202,60 @@ export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext
1923
2202
  updateStatusWidget(ctx);
1924
2203
  return;
1925
2204
  }
1926
- // Ignore stale plans whose file vanished.
1927
2205
  if (!fs.existsSync(snapshot.planPath)) {
1928
2206
  execution = null;
1929
2207
  updateStatusWidget(ctx);
1930
2208
  return;
1931
2209
  }
2210
+ // Re-derive the task tree from the plan file (fresh parse) merged with
2211
+ // the snapshot's persisted statuses — a stale parse cannot freeze progress.
2212
+ let tasks: TaskView[];
2213
+ let items: CheckItem[] = snapshot.items.map((item) => ({ ...item }));
2214
+ let planTasks = snapshot.planTasks;
2215
+ try {
2216
+ const planText = fs.readFileSync(snapshot.planPath, "utf8");
2217
+ planTasks = parsePlanTasks(planText);
2218
+ const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
2219
+ tasks = buildTaskView(planTasks, snapshotProgress);
2220
+ items = parseChecklist(planText);
2221
+ } catch {
2222
+ tasks = snapshot.tasks;
2223
+ }
2224
+ // The snapshot cannot carry an in-flight round (it is memory-only); abort
2225
+ // any live one from the previous session graph and rebuild fresh (CF2-002).
2226
+ abortInFlightReview();
1932
2227
  execution = {
1933
2228
  planPath: snapshot.planPath,
1934
- items: snapshot.items.map((item) => ({ ...item })),
2229
+ items,
2230
+ planTasks,
2231
+ tasks,
2232
+ legacyPlan: snapshot.legacyPlan,
1935
2233
  startedAt: snapshot.startedAt,
1936
2234
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
1937
- implItems: snapshot.implItems ?? [],
1938
- implStatus: { ...(snapshot.implStatus ?? {}) },
1939
- // D-008: chrome language is re-resolved at restore time from the
1940
- // CURRENT config rather than trusted from the snapshot, so a
1941
- // `plans set-language` change survives restarts.
1942
2235
  uiLanguage: resolveUiLanguage(ctx.cwd),
1943
- currentI: snapshot.currentI ?? inferCurrentI(snapshot.implItems, snapshot.items, snapshot.implStatus),
1944
- goalWait: snapshot.goalWait
1945
- ? { ...snapshot.goalWait }
1946
- : { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
2236
+ stall: { ...snapshot.stall, lastSnapshot: null },
2237
+ audit: {
2238
+ rounds: snapshot.audit?.rounds ?? 0,
2239
+ failed: snapshot.audit?.failed ?? [],
2240
+ undeterminable: [],
2241
+ running: false,
2242
+ },
2243
+ review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
2244
+ auditLatch: { auditedThisSettle: false, activity: 0 },
1947
2245
  };
1948
- // Distrust the snapshot's implItems: re-parse + re-lint from the plan
1949
- // file so a stale empty list (older parse or format drift at snapshot
1950
- // time) cannot freeze a fake "I 0/0" panel after a restart.
1951
- try {
1952
- const planText = fs.readFileSync(snapshot.planPath, "utf8");
1953
- execution.implItems = parseImplItems(planText);
1954
- execution.implWarning = lintImplItems(planText);
1955
- } catch {
1956
- /* plan file unreadable mid-restore: keep the snapshot values */
1957
- }
1958
- for (let i = snapshotIndex + 1; i < entries.length; i++) {
1959
- const entry = entries[i];
1960
- if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1961
- execution = null;
1962
- break;
1963
- }
1964
- const message = entry.message;
1965
- if (message && message.role === "assistant") {
1966
- const text = message.content
1967
- .filter((part) => part.type === "text")
1968
- .map((part) => part.text ?? "")
1969
- .join("\n");
1970
- applyDoneMarkers(text);
1971
- applyImplMarkers(text);
1972
- applyCurrentIMarker(text);
1973
- }
1974
- }
1975
- if (execution) {
1976
- resetGoalWaitRuntime(ctx);
1977
- // Rebind the run identity after a restart so the panel's activity row
1978
- // (and any run-status mirroring) resolves to the active run instead of
1979
- // staying null until the next startExecution (CQ1/D-005 wiring gap).
1980
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1981
- executionRunId = active?.run_id ?? null;
1982
- if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1983
- // D-010: replay may have advanced progress past the persisted baseline.
1984
- // Recompute the goal-wait markers; new progress resets the guard counters.
1985
- if (execution.goalWait) {
1986
- const markerSnapshot = goalWaitSnapshot();
1987
- if (markerSnapshot !== execution.goalWait.lastMarkers) {
1988
- execution.goalWait.lastMarkers = markerSnapshot;
1989
- execution.goalWait.noProgressRounds = 0;
1990
- execution.goalWait.waitRounds = 0;
1991
- }
1992
- }
1993
- persist(pi); // refresh snapshot so the next resume has less to rescan
1994
- if (isExecutionComplete()) {
1995
- // Completed during the rescan: restore the planning model on the way out.
1996
- await completeExecution(pi, ctx);
1997
- }
2246
+ execution.stall.lastSnapshot = stallSnapshot();
2247
+ resetContinuationRuntime(ctx);
2248
+ const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
2249
+ executionRunId = active?.run_id ?? null;
2250
+ if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
2251
+ persist(ctx);
2252
+ if (pendingAudit()) {
2253
+ // Self-heal (v0.8): a verifying run with pending checks restarts its
2254
+ // round exactly once per resume — the full predicate (paused, in-flight,
2255
+ // budget) lives inside startReviewRound. Restores never grant budget.
2256
+ latchAuditThisSettle();
2257
+ const chain = launchReviewRound(ctx);
2258
+ if (chain) await chain;
1998
2259
  }
1999
2260
  updateStatusWidget(ctx);
2000
2261
  }