pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
package/src/exec.ts CHANGED
@@ -1,10 +1,18 @@
1
1
  /**
2
- * Plan-execution loop: the tracked execution mode for accepted plans.
2
+ * Plan-execution loop (v0.6.1): the tracked execution mode for accepted
3
+ * plans, driven by the plan's task tree.
3
4
  *
4
5
  * When the user approves the execution handoff, the extension switches into
5
- * execution mode: every agent turn is injected with the remaining verifier
6
- * checklist, assistant messages are scanned for [DONE:VC-xxx] markers, and
7
- * progress is reported through the bottom status bar until every item passes.
6
+ * execution mode: every agent turn is injected with the current wave and
7
+ * remaining tasks, task progress flows in exclusively through the
8
+ * `plans_update_task` tool (status + evidence), the task dashboard shows
9
+ * live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
10
+ * tree), a stall watchdog pauses the run when consecutive rounds produce no
11
+ * task-state change, and when every task reaches a terminal state an
12
+ * independent completion auditor verifies the plan's verification checks —
13
+ * failed checks roll their covered tasks back to pending (audit-flow-only
14
+ * channel), and three failed rounds pause for the user (bounded stopped
15
+ * termination under auto-approve/headless).
8
16
  */
9
17
 
10
18
  import * as fs from "node:fs";
@@ -37,10 +45,9 @@ import {
37
45
  type VccCompactionStats,
38
46
  } from "./compaction.ts";
39
47
  import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
40
- import { execChrome, resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
48
+ import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
41
49
  import { bindRun, resolveActiveRun } from "./run-context.ts";
42
- import { runPiSubagent, type SubagentProgressEvent } from "./subagent.ts";
43
- import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
50
+ import type { SubagentProgressEvent } from "./subagent.ts";
44
51
  import { OwnershipError } from "./run-ownership.ts";
45
52
  import {
46
53
  applyExecutionApproved,
@@ -48,86 +55,80 @@ import {
48
55
  applyExecutionHeadChanged,
49
56
  applyExecutionProgress,
50
57
  applyExecutionStopped,
51
- applyQuestionAsked,
52
- applyReviewRoundStarted,
53
58
  createCheckpoint,
54
59
  loadCheckpoint,
55
60
  mutateCheckpoint,
56
- StaleCheckpointError,
57
61
  planIdentityOf,
58
62
  resolveHeadAt,
59
63
  resolveWorktreeRoot,
60
64
  sha256File,
65
+ StaleCheckpointError,
61
66
  type ExecutionApproval,
67
+ type WorkflowCheckpoint,
62
68
  } from "./workflow-state.ts";
63
69
  import { graphBlockForExecutor } from "./code-graph/prompts.ts";
64
- import {
65
- PANEL_WIDGET_KEY,
66
- derivePanelModel,
67
- deriveNextAction,
68
- deriveImplReviewLoopModel,
69
- formatImplReviewLoopSummaryLine,
70
- formatPanelSummaryLine,
71
- renderImplReviewLoopLines,
72
- renderPanelLines,
73
- themeImplReviewLoopLines,
74
- themePanelLines,
75
- } from "./panel.ts";
76
70
  import { resolveGraphMode } from "./code-graph/mode.ts";
77
71
  import {
78
- TERMINATION_QUESTION,
79
- TERMINATION_OPTIONS,
80
- TERMINATION_RECORDING_INSTRUCTIONS,
81
- defaultImplReviewers,
82
- implReviewerCountPromptLine,
83
- renderTerminationOptions,
84
- } from "./termination-prompt.ts";
85
- import {
86
- extractCoverage,
87
- latestPlanVersion,
88
72
  parseChecklist,
89
- parseImplItems,
90
- resolveImplStatuses,
91
- scanDoneMarkers,
92
- scanImplMarkers,
93
- scanCurrentIMarkers,
94
- resolveCurrentI,
95
- inferCurrentI,
96
- lintImplItems,
73
+ parsePlanTasks,
74
+ flattenTasks,
97
75
  type CheckItem,
98
- type ImplItem,
99
- type ImplMarkerState,
76
+ type PlanTasks,
100
77
  } from "./plan.ts";
78
+ import {
79
+ allTasksTerminal,
80
+ auditRollbackSet,
81
+ auditableChecks,
82
+ buildTaskView,
83
+ currentTask,
84
+ flattenTaskViews,
85
+ taskIsTerminal,
86
+ taskProgress,
87
+ taskProgressMap,
88
+ type TaskProgressMap,
89
+ type TaskView,
90
+ } from "./tasks.ts";
91
+ import { isAutoApproveEnabled as isAutoApproveEnabledLocal } from "./auto-approve.ts";
92
+ import {
93
+ DASHBOARD_WIDGET_KEY,
94
+ deriveDashboardModel,
95
+ formatDashboardSummaryLine,
96
+ formatElapsed,
97
+ renderDashboardLines,
98
+ renderDashboardTreeLines,
99
+ } from "./dashboard.ts";
100
+ import { AUDIT_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit } from "./auditor.ts";
101
+ import { messaging } from "./messaging.ts";
101
102
 
102
103
  export interface ExecState {
103
104
  planPath: string;
105
+ /** Verification checks (VC-###) — the audit's contract. */
104
106
  items: CheckItem[];
107
+ /** Parsed plan task model (kept for re-deriving the view). */
108
+ planTasks: PlanTasks;
109
+ /** Live task tree (the single source of progress). */
110
+ tasks: TaskView[];
111
+ /** True when the plan parsed through the legacy I-### fallback. */
112
+ legacyPlan: boolean;
105
113
  startedAt: string;
106
114
  usage: { inToks: number; outToks: number };
107
- implItems?: ImplItem[];
108
- implStatus?: Record<string, ImplMarkerState>;
109
- /** Plan-lint warning backing the panel's implWarning line. */
110
- implWarning?: string | null;
111
- /** Chrome language for panel/status strings (issue #3); undefined → "en". */
115
+ /** Chrome language for panel/status strings; undefined → "en". */
112
116
  uiLanguage?: UiLanguage;
113
- currentI?: string;
114
- goalWait?: GoalWaitState;
115
- /** v0.6.0: set while a delegated executor child owns the implementation.
116
- * Persisted subset only (modelSelector + startedAt); the AbortController is
117
- * runtime state kept in delegatedRuntime, never persisted. */
118
- delegate?: { modelSelector: string; startedAt: string };
117
+ /** Stall watchdog (v0.6.1): consecutive settled rounds without a task
118
+ * status change; auto-pause at the cap. */
119
+ stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
120
+ /** Completion-audit bookkeeping. */
121
+ audit: { rounds: number; failed: string[]; running: boolean };
122
+ /** Per-settle audit latch (v0.7.1): a settled round fires the completion
123
+ * audit at most once, so the turn_end / agent_before_settle / resume entry
124
+ * points cannot double-consume a round when several land in one settle.
125
+ * Created on demand by auditLatchOf(); every construction path may omit it. */
126
+ auditLatch?: { auditedThisSettle: boolean; activity: number };
119
127
  }
120
128
 
121
129
  /**
122
- * D-008 (issue #3): re-resolve the chrome language and repaint the panel and
123
- * status bar. Called by the plans tool right after a successful
124
- * `set-language` so an executing run switches language without a restart
125
- * (and without reading config on every render tick).
126
- *
127
- * Capability guard (implementation review F-001): partial contexts (some
128
- * command/test harnesses expose only notify/select/input) may lack
129
- * setStatus/theme — the refresh must stay a no-op there instead of throwing
130
- * into the caller's error path (updateStatusWidget assumes a full ui).
130
+ * D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
131
+ * and status bar right after a `set-language` change.
131
132
  */
132
133
  export function refreshUiLanguage(ctx: ExtensionContext): void {
133
134
  if (execution) execution.uiLanguage = resolveUiLanguage(ctx.cwd);
@@ -135,43 +136,35 @@ export function refreshUiLanguage(ctx: ExtensionContext): void {
135
136
  updateStatusWidget(ctx);
136
137
  }
137
138
 
138
- export interface GoalWaitState {
139
- noProgressRounds: number;
140
- waitRounds: number;
141
- /** Marker/progress snapshot of the last goal-wait round; null = baseline not set. */
142
- lastMarkers: string | null;
143
- paused: boolean;
144
- pausedReason?: string;
145
- }
146
-
147
- const GOAL_WAIT_MAX_NO_PROGRESS = 3;
148
- const GOAL_WAIT_MAX_WAITING = 6;
139
+ /** Consecutive no-progress rounds before the watchdog pauses (D-021). */
140
+ const STALL_MAX_ROUNDS = 3;
149
141
 
150
142
  let execution: ExecState | null = null;
151
143
 
152
- export const GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
144
+ export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
145
+ /** Legacy v0.6.0 continuation message type — filtered on restore. */
146
+ const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
153
147
 
154
- interface GoalWaitRuntime {
148
+ interface ContinuationRuntime {
155
149
  owner: ExecState;
156
150
  session: ExtensionContext["sessionManager"];
157
151
  handled: boolean;
158
152
  stopReason?: string;
159
- text: string;
160
153
  wakeId?: string;
161
154
  }
162
155
 
163
- // Dispatch identity belongs to a live session, never to a persisted checklist.
164
- let goalWaitRuntime: GoalWaitRuntime | null = null;
156
+ // Dispatch identity belongs to a live session, never to a persisted state.
157
+ let continuationRuntime: ContinuationRuntime | null = null;
165
158
 
166
- function resetGoalWaitRuntime(ctx: ExtensionContext): void {
167
- goalWaitRuntime = execution
168
- ? { owner: execution, session: ctx.sessionManager, handled: false, text: "" }
159
+ function resetContinuationRuntime(ctx: ExtensionContext): void {
160
+ continuationRuntime = execution
161
+ ? { owner: execution, session: ctx.sessionManager, handled: false }
169
162
  : null;
170
163
  }
171
164
 
172
- function currentGoalWaitRuntime(ctx: ExtensionContext): GoalWaitRuntime | null {
173
- return goalWaitRuntime?.owner === execution && goalWaitRuntime.session === ctx.sessionManager
174
- ? goalWaitRuntime
165
+ function currentContinuationRuntime(ctx: ExtensionContext): ContinuationRuntime | null {
166
+ return continuationRuntime?.owner === execution && continuationRuntime.session === ctx.sessionManager
167
+ ? continuationRuntime
175
168
  : null;
176
169
  }
177
170
 
@@ -185,17 +178,14 @@ export function consumePendingExecutionFlush(): boolean {
185
178
  return pending;
186
179
  }
187
180
 
188
- function requestExecutionFlush(_pi: ExtensionAPI, _ctx: ExtensionContext): void {
189
- // Unconditional defer. turn_end fires mid-run in a gap between agent
190
- // operations where isIdle() reads true; persistence happens only at the
191
- // drain points: agent_settled, the next before_agent_start, and stop/complete.
181
+ function requestExecutionFlush(): void {
192
182
  pendingExecutionFlush = true;
193
183
  }
194
184
 
195
- export function drainExecutionFlush(pi: ExtensionAPI, ctx: ExtensionContext): void {
185
+ export function drainExecutionFlush(ctx: ExtensionContext): void {
196
186
  if (!execution || !pendingExecutionFlush) return;
197
187
  pendingExecutionFlush = false;
198
- persist(pi);
188
+ persist(ctx);
199
189
  updateStatusWidget(ctx);
200
190
  }
201
191
 
@@ -209,19 +199,22 @@ export interface CheckpointExecutionLoad {
209
199
  doneVcIds?: string[];
210
200
  reverifyAll?: boolean;
211
201
  pausedReason?: string;
202
+ /** v0.6.1: true when the checkpoint parsed through the legacy fallback. */
203
+ legacyPlan?: boolean;
204
+ /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
205
+ * detected — resume requires a fresh handoff approval. */
206
+ legacyDelegate?: boolean;
212
207
  error?: string;
213
208
  }
214
209
 
215
210
  /**
216
- * Shared restore primitive (I-005/I-006): load the executing state from a run
217
- * checkpoint into THIS session. Authorization is kept only when the recorded
218
- * approval matches the current plan digest; a HEAD change keeps the
219
- * authorization but re-verifies previously verified VCs (D-011/F-001).
220
- * F-002: the loaded state is persisted to the current session IMMEDIATELY so
221
- * session_start/session_tree restore paths cannot silently clear it.
211
+ * Shared restore primitive: load the executing state from a run checkpoint
212
+ * into THIS session. Authorization is kept only when the recorded approval
213
+ * matches the current plan digest; a HEAD change keeps the authorization but
214
+ * re-opens previously closed tasks (D-023: reverifyAll → task statuses are
215
+ * dropped and re-run).
222
216
  */
223
217
  export function loadExecutionFromCheckpoint(
224
- pi: ExtensionAPI,
225
218
  ctx: ExtensionContext,
226
219
  runId: string,
227
220
  ): CheckpointExecutionLoad {
@@ -235,9 +228,6 @@ export function loadExecutionFromCheckpoint(
235
228
  return { status: "plan-missing", error: planPath ? `plan file vanished: ${planPath}` : "checkpoint has no plan identity" };
236
229
  }
237
230
  const planText = fs.readFileSync(planPath, "utf8");
238
- // F-002 (implementation review): the recorded plan identity is over BYTES —
239
- // an in-place edit at the same path must not inherit the authorization or
240
- // the verified VCs. Refuse the load and require a fresh handoff.
241
231
  if (sha256File(planPath) !== cp.plan.sha256) {
242
232
  return {
243
233
  status: "plan-mismatch" as const,
@@ -246,74 +236,107 @@ export function loadExecutionFromCheckpoint(
246
236
  }
247
237
  const items = parseChecklist(planText);
248
238
  if (items.length === 0) {
249
- return { status: "plan-missing", error: `${planPath} has no parsable verifier checklist` };
239
+ return { status: "plan-missing", error: `${planPath} has no parsable verification checks` };
240
+ }
241
+ const planTasks = parsePlanTasks(planText);
242
+ if (planTasks.tasks.length === 0) {
243
+ return { status: "plan-missing", error: `${planPath} has no parsable tasks (## Tasks or legacy ## Implementation Items)` };
250
244
  }
251
- const implItems = parseImplItems(planText);
252
- const doneIds = new Set(cp.execution.doneVcIds);
253
- // D-011/F-001: an unchanged plan digest keeps the recorded authorization;
254
- // a changed HEAD under it forces re-verification of previously verified VCs.
255
- // F-006 (implementation review): an approval without a resolvable HEAD
256
- // recorded an unverifiable code state — re-verify instead of trusting.
257
245
  const headNow = resolveHeadAt(ctx.cwd);
258
246
  const headUnverifiable = cp.execution.approval === null || cp.execution.approval.headAtApproval === null;
259
247
  const headChanged =
260
248
  cp.execution.approval !== null &&
261
249
  cp.execution.approval.headAtApproval !== null &&
262
250
  cp.execution.approval.headAtApproval !== headNow;
251
+ // D-023: reverifyAll re-opens every closed task (statuses dropped); a
252
+ // normal restore replays the persisted task progress map.
263
253
  const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
254
+ const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
255
+ const tasks = buildTaskView(planTasks, progress);
256
+ // An audit-cap pause grants a fresh audit budget on restore (mirrors
257
+ // resumeGoalWaitIfPaused) so the first turn can actually re-audit.
258
+ const wasAuditCapPause = (cp.execution.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
264
259
  if (!reverifyAll) {
265
- for (const item of items) {
266
- if (doneIds.has(item.id)) item.done = true;
260
+ for (const id of cp.execution.doneVcIds) {
261
+ const item = items.find((candidate) => candidate.id === id);
262
+ if (item) item.done = true;
267
263
  }
268
264
  }
269
265
  execution = {
270
266
  planPath,
271
267
  items,
268
+ planTasks,
269
+ tasks,
270
+ legacyPlan: planTasks.legacy,
272
271
  startedAt: utcNow(),
273
272
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
274
- implItems,
275
- implStatus: { ...cp.execution.implStatus },
276
- implWarning: lintImplItems(planText),
277
273
  uiLanguage: resolveUiLanguage(ctx.cwd),
278
- currentI: cp.execution.currentI,
279
- goalWait: {
280
- noProgressRounds: 0,
281
- waitRounds: 0,
282
- lastMarkers: null,
283
- paused: cp.execution.pausedReason !== undefined,
284
- pausedReason: cp.execution.pausedReason,
274
+ // D-020: a paused legacy (or stopped) execution rebuilds unpaused —
275
+ // the resume itself is the user's intent; the reason is surfaced in
276
+ // the resume brief instead. An audit-cap pause additionally grants a
277
+ // fresh audit budget here (mirrors resumeGoalWaitIfPaused), so the
278
+ // first turn after a cross-session resume can actually re-audit.
279
+ stall: {
280
+ rounds: 0,
281
+ lastSnapshot: null,
282
+ paused: false,
283
+ pausedReason: undefined,
284
+ },
285
+ audit: {
286
+ rounds: wasAuditCapPause ? 0 : (cp.execution.audit?.rounds ?? 0),
287
+ failed: [],
288
+ running: false,
285
289
  },
290
+ auditLatch: { auditedThisSettle: false, activity: 0 },
286
291
  };
292
+ execution.stall.lastSnapshot = stallSnapshot();
287
293
  executionRunId = runId;
288
294
  bindRun(ctx.sessionManager, ctx.cwd, runId);
289
- resetGoalWaitRuntime(ctx);
290
- pendingExecutionFlush = false; // restored state: no inherited flush debt
295
+ resetContinuationRuntime(ctx);
296
+ pendingExecutionFlush = false;
291
297
  resetExecutionCompactionState(ctx);
292
- // v0.6.0 orphaned-delegate detection: a restart never carries a live child.
293
- // If the checkpoint/session carries delegate state, surface it so the user
294
- // knows the previous executor died mid-run (VC state is intact, resumable).
295
- if (cp.execution.delegate) {
298
+ if (wasAuditCapPause) {
299
+ withExecutionCheckpoint(ctx, (cp2) =>
300
+ applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
301
+ );
302
+ }
303
+ // D-020: an orphaned v0.6.0 delegated executor never survives a restart.
304
+ // Its checkpoint delegate marker REFUSES the direct load — the run must
305
+ // re-enter through the execution handoff so the C-006 approval gate
306
+ // applies; execution restarts from the first task after re-approval.
307
+ const legacyDelegate = cp.execution.delegate !== undefined;
308
+ if (legacyDelegate) {
296
309
  try {
297
310
  ctx.ui.notify?.(
298
- `pi-plans: the previous delegated executor (${cp.execution.delegate.modelSelector}) did not finish before this session ended. Verified VC state is preserved; resume with /plans-execute.`,
311
+ "pi-plans: this run was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Re-approve via /plans-execute; execution restarts from the first task (the 0.6.0 progress record cannot map onto the task tree).",
299
312
  "warning",
300
313
  );
301
314
  } catch {
302
- /* notification is best-effort */
315
+ /* best-effort */
303
316
  }
317
+ // Refuse the load: no execution state may activate without the fresh
318
+ // C-006 handoff approval.
319
+ execution = null;
320
+ executionRunId = null;
321
+ return {
322
+ status: "no-execution",
323
+ legacyDelegate: true,
324
+ error: "orphaned v0.6.0 delegated executor; re-approve via /plans-execute (execution restarts from the first task)",
325
+ };
304
326
  }
305
327
  if (headChanged) {
306
328
  withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
307
329
  }
308
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
309
- persist(pi); // F-002: immediate session snapshot
330
+ persist(ctx);
310
331
  updateStatusWidget(ctx);
311
332
  return {
312
333
  status: "loaded",
313
334
  planPath,
314
- doneVcIds: [...doneIds],
335
+ doneVcIds: [...cp.execution.doneVcIds],
315
336
  reverifyAll,
316
337
  pausedReason: cp.execution.pausedReason,
338
+ legacyPlan: planTasks.legacy,
339
+ legacyDelegate,
317
340
  };
318
341
  }
319
342
 
@@ -327,7 +350,6 @@ interface ExecutionCompactionState {
327
350
  lastSuccessfulUsagePercent: number | null;
328
351
  lastSuccessfulAt: string | null;
329
352
  rearmPending: boolean;
330
- /** Terminal failure metadata is retained for diagnostics, not proactive retry. */
331
353
  terminalBackoffTokens: number | null;
332
354
  pendingStats: VccCompactionStats | null;
333
355
  pendingFollowUpPrompt: string | null;
@@ -387,63 +409,21 @@ export function handleExecutionTurnCompaction(ctx: ExtensionContext): void {
387
409
  consumeExecutionCompactionResumeGuard(ctx);
388
410
  }
389
411
 
390
- export function computeExecutionProgress(execution: ExecState): { done: number; total: number } {
391
- const implItems = execution.implItems ?? [];
392
- if (implItems.length) {
393
- const statuses = resolveImplStatuses(implItems, execution.items, execution.implStatus);
394
- const counted = implItems.filter((impl) =>
395
- execution.items.some((item) => extractCoverage(item.text).includes(impl.id)),
396
- );
397
- const total = counted.length > 0 ? counted.length : implItems.length;
398
- const vcDone = counted.filter((impl) => statuses[impl.id] === "vc-passed").length;
399
- const currentIndex = execution.currentI
400
- ? implItems.findIndex((impl) => impl.id === execution.currentI)
401
- : -1;
402
- return {
403
- done: Math.min(total, Math.max(vcDone, currentIndex < 0 ? 0 : currentIndex)),
404
- total,
405
- };
406
- }
407
- return {
408
- done: execution.items.filter((item) => item.done).length,
409
- total: execution.items.length,
410
- };
411
- }
412
-
413
- function formatElapsed(startedAt: string): string {
414
- const total = Math.max(0, Math.floor((Date.now() - Date.parse(startedAt)) / 1000));
415
- const h = String(Math.floor(total / 3600)).padStart(2, "0");
416
- const m = String(Math.floor((total % 3600) / 60)).padStart(2, "0");
417
- const sec = String(total % 60).padStart(2, "0");
418
- return `${h}:${m}:${sec}`;
419
- }
420
-
421
412
  function formatToks(tokens: number): string {
422
413
  const n = Math.max(0, Math.round(tokens));
423
414
  return n < 1000 ? String(n) : `${(n / 1000).toFixed(1)}k`;
424
415
  }
425
416
 
426
417
  export function formatExecutionStatusLine(execution: ExecState): string {
427
- const progress = computeExecutionProgress(execution);
428
- let line = `⌛ plans ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
429
- const goalWait = execution.goalWait;
430
- if (goalWait?.paused) {
431
- line += ` · ⏸ goal-wait paused (${goalWait.pausedReason ?? "paused"})`;
432
- } else if (goalWait && (goalWait.noProgressRounds > 0 || goalWait.waitRounds > 0)) {
433
- line += execChrome(execution.uiLanguage ?? "en").goalWait(goalWait.noProgressRounds, goalWait.waitRounds);
418
+ const progress = taskProgress(execution.tasks);
419
+ let line = `⌛ plans tasks ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
420
+ if (execution.stall.paused) {
421
+ line += ` · ⏸ paused (${execution.stall.pausedReason ?? "stalled"})`;
434
422
  }
435
423
  return line;
436
424
  }
437
425
 
438
- /** Approximation for "a subprocess is pending" (matches the exec loop's
439
- * `/waiting for/` backoff heuristic; F-008). Passed explicitly into the
440
- * shared model so the panel, status line and injection text agree. */
441
- export function executionIsWaiting(execution: ExecState): boolean {
442
- const gw = execution.goalWait;
443
- return gw !== undefined && !gw.paused && gw.waitRounds > 0;
444
- }
445
-
446
- /** Resolve the run topic for panel headers (falls back to the run id). */
426
+ /** Resolve the run topic for dashboard headers (falls back to the run id). */
447
427
  function panelTopic(ctx: ExtensionContext): string {
448
428
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
449
429
  if (active) {
@@ -454,155 +434,77 @@ function panelTopic(ctx: ExtensionContext): string {
454
434
  return "pi-plans";
455
435
  }
456
436
 
457
- /** Run info for the panel activity line (CQ1/D-005). Read by the in-flight
458
- * executionRunId so a stale/missing run record degrades to null (the model
459
- * then falls back to the bare phase word) instead of showing another run's
460
- * status. */
461
- function panelRunInfo(ctx: ExtensionContext): { status: string; created_at: string; updated_at: string } | null {
462
- if (!executionRunId) return null;
463
- // D-005 pointer-consistency: if the workdir's active pointer has moved to
464
- // another run (second session / external CLI mutation) while this
465
- // execution is live, the activity row degrades to the phase word rather
466
- // than mixing the new active run's topic (header) with the old run's
467
- // status (impl-review r1 F-001).
468
- if (resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id !== executionRunId) return null;
469
- const run = getRun(ctx.cwd, executionRunId);
470
- if (!run) return null;
471
- return { status: run.status, created_at: run.created_at, updated_at: run.updated_at };
472
- }
437
+ let dashboardRegistered = false;
438
+ let dashboardExpanded = false;
473
439
 
474
- let panelRegistered = false;
475
- let loopPanelRegistered = false;
440
+ /** Toggle the dashboard's expanded tree view (Ctrl+Shift+T). */
441
+ export function toggleDashboardExpanded(ctx: ExtensionContext): void {
442
+ dashboardExpanded = !dashboardExpanded;
443
+ dashboardRegistered = false; // force re-registration with the new mode
444
+ updateStatusWidget(ctx);
445
+ }
476
446
 
477
- /** Live implementation-review loop state for the panel: the widget stays
478
- * alive while the active run is done BUT its checkpoint is still in the
479
- * implementation-review phase (D-3). Read fresh on every call so post-write
480
- * redraws (index.ts turn-end updateStatusWidget) never show stale rounds
481
- * (D-8). Returns null once the phase flips to completed. */
482
- function implReviewLoopState(
483
- ctx: ExtensionContext,
484
- ): { topic: string; review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } } | null {
485
- // v0.6.0: resolveActiveRun only returns NON-TERMINAL runs, but this state
486
- // is by definition attached to a DONE run — fall back to the newest run of
487
- // any status (display-only) when the active resolution is null.
488
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd) ?? (() => {
489
- const latest = latestRun(ctx.cwd);
490
- return latest === null ? null : { run_id: latest.run_id, artifact_dir: latest.artifact_dir };
491
- })();
492
- if (!active) return null;
493
- if (getRun(ctx.cwd, active.run_id)?.status !== "done") return null;
494
- const load = loadCheckpoint(ctx.cwd, active.run_id);
495
- if (load.status !== "ok" || load.checkpoint.phase !== "implementation-review") return null;
496
- return { topic: panelTopic(ctx), review: load.checkpoint.implementationReview };
447
+ export function isDashboardExpanded(): boolean {
448
+ return dashboardExpanded;
497
449
  }
498
450
 
499
451
  /**
500
- * Register/update the fixed tasks' status panel (aboveEditor widget) for the
501
- * current execution, or unregister it when execution is gone. The panel and
502
- * the bottom status line share the same pure model (D-003/D-014/D-015). The
503
- * widget uses the factory form and reads the LIVE theme via `ui.theme` inside
504
- * render(width) — per-line width math happens on plain text first, then the
505
- * current theme is applied, so theme hot-swaps and resize never produce stale
506
- * colors or wrapped rows (F-004).
452
+ * Register/update the task dashboard (aboveEditor widget) for the current
453
+ * execution, or unregister it when execution is gone. The compact and the
454
+ * expanded tree view share one widget key so they never stack.
507
455
  */
508
456
  function updatePanelWidget(ctx: ExtensionContext): void {
509
- // Capability guard: older Pi hosts and test harness mocks may not expose
510
- // setWidget (the panel is a UI nicety, never a correctness dependency).
511
457
  if (!ctx.hasUI || typeof ctx.ui.setWidget !== "function") return;
512
- // Implementation-review loop widget (D-3): keeps the panel alive after
513
- // execution ends while the loop is live; unregisters when the phase
514
- // completes. Lives under the same widget key so the two widgets never
515
- // stack.
516
- const loop = execution === null ? implReviewLoopState(ctx) : null;
517
- if (execution === null && loop) {
518
- if (!loopPanelRegistered) {
519
- ctx.ui.setWidget(
520
- PANEL_WIDGET_KEY,
521
- (ui, _theme) => ({
522
- render(width: number) {
523
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
524
- // D-8 freshness: re-read the loop state per render; a phase flip to
525
- // completed renders an empty box until the next turn-end refresh
526
- // unregisters it (index.ts always calls updateStatusWidget then).
527
- const live = implReviewLoopState(ctx);
528
- if (!live) return [];
529
- const model = deriveImplReviewLoopModel(live.topic, live.review);
530
- const lines = renderImplReviewLoopLines(model, width);
531
- return theme ? themeImplReviewLoopLines(lines, theme as never) : lines;
532
- },
533
- }),
534
- { placement: "aboveEditor" },
535
- );
536
- loopPanelRegistered = true;
537
- }
538
- } else if (loopPanelRegistered) {
539
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
540
- loopPanelRegistered = false;
541
- }
542
458
  if (!execution) {
543
- if (panelRegistered) {
544
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
545
- panelRegistered = false;
459
+ if (dashboardRegistered) {
460
+ ctx.ui.setWidget(DASHBOARD_WIDGET_KEY, undefined);
461
+ dashboardRegistered = false;
546
462
  }
547
463
  return;
548
464
  }
549
- if (!panelRegistered) {
550
- ctx.ui.setWidget(
551
- PANEL_WIDGET_KEY,
552
- (ui, _theme) => ({
553
- render(width: number) {
554
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
555
- const current = execution;
556
- if (!current) return [];
557
- // F-004 (impl review r1): recompute the topic per render so a
558
- // cross-run restart without an intervening unregister cannot
559
- // show a stale box header.
560
- const model = derivePanelModel(current, panelTopic(ctx), executionIsWaiting(current), panelRunInfo(ctx));
561
- const lines = renderPanelLines(model, width);
562
- // Uniform-gray frame: │ borders never inherit the line color;
563
- // accents live between the borders only (themePanelLines).
564
- return theme ? themePanelLines(lines, model, theme) : lines;
565
- },
566
- }),
567
- { placement: "aboveEditor" },
568
- );
569
- panelRegistered = true;
570
- } else {
571
- // Factory components are re-created on every registration; content
572
- // updates flow through the closure reads at render time, so a no-op
573
- // re-set is unnecessary. Trigger one re-render via a cheap status
574
- // touch is NOT used — event-driven flush only (D-013).
575
- }
465
+ if (dashboardRegistered) return;
466
+ ctx.ui.setWidget(
467
+ DASHBOARD_WIDGET_KEY,
468
+ (ui, _theme) => ({
469
+ render(width: number) {
470
+ const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
471
+ const current = execution;
472
+ if (!current) return [];
473
+ const model = deriveDashboardModel(panelTopic(ctx), current.tasks, current.items, {
474
+ paused: current.stall.paused,
475
+ pausedReason: current.stall.pausedReason,
476
+ auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
477
+ auditFailed: current.audit.failed,
478
+ startedAt: current.startedAt,
479
+ usage: current.usage,
480
+ });
481
+ const lines = dashboardExpanded
482
+ ? renderDashboardTreeLines(model, width, theme)
483
+ : renderDashboardLines(model, width, theme);
484
+ return lines;
485
+ },
486
+ }),
487
+ { placement: "aboveEditor" },
488
+ );
489
+ dashboardRegistered = true;
576
490
  }
577
491
 
578
492
  export function updateStatusWidget(ctx: ExtensionContext): void {
579
493
  updatePanelWidget(ctx);
580
- if (execution) {
581
- // D-015: the status line derives from the same panel model.
582
- const line = formatPanelSummaryLine(
583
- derivePanelModel(execution, panelTopic(ctx), executionIsWaiting(execution), panelRunInfo(ctx)),
584
- );
585
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
494
+ if (execution && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
495
+ const model = deriveDashboardModel(panelTopic(ctx), execution.tasks, execution.items, {
496
+ paused: execution.stall.paused,
497
+ pausedReason: execution.stall.pausedReason,
498
+ auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
499
+ auditFailed: execution.audit.failed,
500
+ });
501
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
586
502
  return;
587
503
  }
588
504
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
589
- if (active) {
590
- // Idle indicator depends on the run's lifecycle, not just its existence:
591
- // done reads as finished, abandoned as closed, stopped/accepted as paused.
592
- // v0.6.0: resolveActiveRun only returns NON-TERMINAL runs; for the pure
593
- // display line below, fall back to the newest run of any status so a
594
- // finished workdir still shows its last run's outcome.
505
+ if (active && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
595
506
  const status = getRun(ctx.cwd, active.run_id)?.status ?? latestRun(ctx.cwd)?.status;
596
507
  if (status === "done") {
597
- // D-3/D-015: while the implementation-review loop is live (checkpoint
598
- // still in the implementation-review phase), the status line mirrors
599
- // the loop box model instead of a bare "(done)".
600
- const loop = implReviewLoopState(ctx);
601
- if (loop) {
602
- const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
603
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
604
- return;
605
- }
606
508
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${active.run_id} (done)`));
607
509
  return;
608
510
  }
@@ -618,63 +520,63 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
618
520
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("warning", `⌛ plans: ${active.run_id}`));
619
521
  return;
620
522
  }
523
+ if (status === "executing") {
524
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
525
+ return;
526
+ }
621
527
  if (status === "planning") {
622
- // Planning phase: 💬 while still in Q&A, 📝 once a PLAN draft exists
623
- // — kept until execution starts (then ⌛ takes over).
624
- const emoji = latestPlanVersion(active.artifact_dir) ? "📝" : "💬";
528
+ const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
625
529
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
626
530
  return;
627
531
  }
628
- // unknown status: no indicator.
629
532
  }
630
- // Terminal-only workdir: resolveActiveRun is null, but the display line
631
- // still reports the newest run's outcome (done/abandoned, plus its live
632
- // implementation-review loop).
633
- const terminalLatest = latestRun(ctx.cwd);
634
- if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
635
- if (terminalLatest.status === "done") {
636
- const loop = implReviewLoopState(ctx);
637
- if (loop) {
638
- const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
639
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
640
- return;
533
+ if (typeof ctx.ui?.setStatus === "function" && ctx.ui.theme) {
534
+ const terminalLatest = latestRun(ctx.cwd);
535
+ if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
536
+ if (terminalLatest.status === "done") {
537
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
538
+ } else {
539
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
641
540
  }
642
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
643
- } else {
644
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
541
+ return;
645
542
  }
646
- return;
543
+ ctx.ui.setStatus("pi-plans", undefined);
647
544
  }
648
- ctx.ui.setStatus("pi-plans", undefined);
649
545
  }
650
546
 
651
- function persist(pi: ExtensionAPI): void {
547
+ function parseLatestPlanExists(artifactDir: string): boolean {
548
+ try {
549
+ const names = fs.readdirSync(artifactDir);
550
+ return names.some((name) => /^PLAN_v\d+\.(md|markdown)$/i.test(name));
551
+ } catch {
552
+ return false;
553
+ }
554
+ }
555
+
556
+ function persist(ctx: ExtensionContext): void {
652
557
  if (!execution) return;
653
- pi.appendEntry("pi-plans-exec", {
558
+ messaging().appendEntry("pi-plans-exec", {
654
559
  planPath: execution.planPath,
655
560
  items: execution.items,
561
+ planTasks: execution.planTasks,
562
+ tasks: execution.tasks,
563
+ legacyPlan: execution.legacyPlan,
656
564
  startedAt: execution.startedAt,
657
565
  usage: execution.usage,
658
- implItems: execution.implItems,
659
- implStatus: execution.implStatus,
660
- implWarning: execution.implWarning ?? null,
661
- currentI: execution.currentI,
662
- goalWait: execution.goalWait,
663
- delegate: execution.delegate ?? null,
566
+ stall: execution.stall,
567
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
664
568
  });
665
569
  }
666
570
 
667
571
  /** Checkpoint bookkeeping for the executing run; best-effort for legacy runs
668
- * without checkpoints (their cross-session resume degrades to R-008 rules). */
669
- function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./workflow-state.ts").WorkflowCheckpoint) => import("./workflow-state.ts").WorkflowCheckpoint): void {
572
+ * without checkpoints. */
573
+ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: WorkflowCheckpoint) => WorkflowCheckpoint): void {
670
574
  if (!execution) return;
671
575
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
672
576
  if (!active || active.run_id !== executionRunId) return;
673
577
  try {
674
578
  mutateCheckpoint(ctx.cwd, active.run_id, mutator);
675
579
  } catch (error) {
676
- // F-005 (implementation review): ownership loss and revision staleness
677
- // must stop the advance, not vanish into the catch block.
678
580
  if (error instanceof OwnershipError || error instanceof StaleCheckpointError) throw error;
679
581
  /* legacy run or corrupt checkpoint: session snapshot still carries the loop */
680
582
  }
@@ -682,284 +584,104 @@ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./
682
584
 
683
585
  let executionRunId: string | null = null;
684
586
 
587
+ export interface StartExecutionInput {
588
+ planPath: string;
589
+ planTasks: PlanTasks;
590
+ items: CheckItem[];
591
+ }
592
+
685
593
  export async function startExecution(
686
- pi: ExtensionAPI,
687
594
  ctx: ExtensionContext,
688
- planPath: string,
689
- items: CheckItem[],
690
- implItems?: ImplItem[],
691
- opts?: StartExecutionOptions,
595
+ input: StartExecutionInput,
692
596
  ): Promise<void> {
597
+ const tasks = buildTaskView(input.planTasks);
693
598
  execution = {
694
- planPath,
695
- items,
599
+ planPath: input.planPath,
600
+ items: input.items,
601
+ planTasks: input.planTasks,
602
+ tasks,
603
+ legacyPlan: input.planTasks.legacy,
696
604
  startedAt: utcNow(),
697
605
  usage: { inToks: 0, outToks: 0 },
698
- implItems: implItems ?? [],
699
- implStatus: {},
700
606
  uiLanguage: resolveUiLanguage(ctx.cwd),
701
- // F-001 (impl review r1): derive the plan-lint warning on the live
702
- // handoff path too, so the panel shows the ⚠ line immediately for a
703
- // zero-parse section instead of only after a checkpoint restore.
704
- implWarning: (() => {
705
- try {
706
- return lintImplItems(fs.readFileSync(planPath, "utf8"));
707
- } catch {
708
- return null;
709
- }
710
- })(),
711
- goalWait: { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
607
+ stall: { rounds: 0, lastSnapshot: null, paused: false },
608
+ audit: { rounds: 0, failed: [], running: false },
609
+ auditLatch: { auditedThisSettle: false, activity: 0 },
712
610
  };
713
- // Seed the marker baseline so the first quiet round is counted against a
714
- // real snapshot instead of counting unconditionally (F-006).
715
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
716
- resetGoalWaitRuntime(ctx);
717
- pendingExecutionFlush = false; // fresh run: no inherited flush debt
611
+ // Seed the watchdog baseline only after `execution` points at the new state
612
+ // (stallSnapshot reads the live execution).
613
+ execution.stall.lastSnapshot = stallSnapshot();
614
+ resetContinuationRuntime(ctx);
615
+ pendingExecutionFlush = false;
718
616
  resetExecutionCompactionState(ctx);
719
- persist(pi);
617
+ persist(ctx);
720
618
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
721
619
  executionRunId = active?.run_id ?? null;
722
620
  if (active) {
723
621
  bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
724
- // I-005: durable approval evidence — run + plan digest + HEAD at
725
- // approval (D-003/D-011). Sets phase executing via the state machine.
726
622
  try {
727
623
  const load = loadCheckpoint(ctx.cwd, active.run_id);
728
624
  if (load.status === "missing") {
729
625
  createCheckpoint(ctx.cwd, { runId: active.run_id, originWorkdir: ctx.cwd, workdir: ctx.cwd });
730
626
  }
731
627
  const approval: ExecutionApproval = {
732
- plan: planIdentityOf(path.resolve(planPath), 1),
628
+ plan: planIdentityOf(path.resolve(input.planPath), 1),
733
629
  worktree: resolveWorktreeRoot(ctx.cwd) ?? path.resolve(ctx.cwd),
734
630
  headAtApproval: resolveHeadAt(ctx.cwd),
735
631
  approvedAt: utcNow(),
736
632
  };
737
633
  mutateCheckpoint(ctx.cwd, active.run_id, (cp) => {
738
- // Plan refinement may not have recorded the plan identity yet.
739
634
  const withPlan = cp.plan === null ? { ...cp, plan: approval.plan } : cp;
740
- // The checkpoint may not carry accept-execute (legacy flow);
741
- // approval here came from the explicit handoff confirmation.
742
635
  const aligned = withPlan.nextAction === "accept-execute"
743
636
  ? withPlan
744
637
  : { ...withPlan, nextAction: "accept-execute" as const };
745
638
  return applyExecutionApproved(aligned, approval);
746
639
  });
747
- // Plan-lint entry point (execute handoff): a plan whose Implementation
748
- // Items section parses to zero items gets a durable run notice so the
749
- // execution panel's warning is backed by persisted evidence.
750
- lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(planPath));
640
+ lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(input.planPath));
751
641
  } catch (error) {
752
- // F-002 (implementation review): a plan-digest mismatch between the
753
- // recorded checkpoint plan and the approval must fail closed and
754
- // visibly — never silently execute without durable approval.
755
642
  if (error instanceof StateError && /does not match/.test(error.message)) throw error;
756
643
  /* legacy/corrupt checkpoint: run status still transitions below */
757
644
  }
758
645
  try {
759
646
  setRunStatus(ctx.cwd, active.run_id, "executing");
760
647
  } catch {
761
- /* status bookkeeping is best-effort */
648
+ /* best-effort */
762
649
  }
763
650
  }
764
- pi.sendMessage(
651
+ const progress = taskProgress(tasks);
652
+ messaging().sendMessage(
765
653
  {
766
654
  customType: "pi-plans-exec-start",
767
- content: `**pi-plans: executing** \`${planPath}\` — ${items.length} verifier item(s). Progress appears in the bottom status bar; mark verified items with \`[DONE:VC-xxx]\`.`,
655
+ content: `**pi-plans: executing** \`${input.planPath}\` — ${progress.total} task(s) in ${input.planTasks.legacy ? "legacy" : "task-tree"} mode, ${input.items.length} verification check(s). Report progress with the \`plans_update_task\` tool; the dashboard tracks every task (Ctrl+Shift+T expands the tree).`,
768
656
  display: true,
769
657
  },
770
658
  { triggerTurn: false },
771
659
  );
772
- // Delegated runtime (v0.6.0): one executor child implements the whole plan
773
- // while this tool call blocks; the parent mirrors VC progress from the
774
- // child's streamed assistant messages (R-9..R-12).
775
660
  updateStatusWidget(ctx);
776
- if (opts?.runtime && opts.runtime !== "current-session") {
777
- await runDelegatedExecution(pi, ctx, opts.runtime.modelSelector, opts.signal);
778
- }
779
- }
780
-
781
- /** Where a chosen execution runs: this session, or a delegated executor child. */
782
- export type ExecutionRuntime = "current-session" | { modelSelector: string };
783
-
784
- export interface StartExecutionOptions {
785
- /** "current-session" (default) or a delegated executor model selector. */
786
- runtime?: ExecutionRuntime;
787
- /** Tool-call abort signal, threaded into the delegated child. */
788
- signal?: AbortSignal;
789
- }
790
-
791
- /** Default delegated executor timeout when config omits executor_timeout_minutes. */
792
- const DELEGATE_DEFAULT_TIMEOUT_MINUTES = 60;
793
- void DELEGATE_DEFAULT_TIMEOUT_MINUTES;
794
-
795
- /** Live AbortController for the delegated executor child (runtime-only state). */
796
- let delegatedRuntime: AbortController | null = null;
797
-
798
- /** Abort the delegated executor child, if one is running (used by /plans-stop). */
799
- export function abortDelegatedExecutor(): boolean {
800
- if (delegatedRuntime === null) return false;
801
- delegatedRuntime.abort();
802
- return true;
803
- }
804
-
805
- function executorAgentPrompt(): string {
806
- try {
807
- const agentPath = new URL("../agents/executor.md", import.meta.url);
808
- return fs.readFileSync(agentPath, "utf8");
809
- } catch {
810
- return "You are a delegated plan executor in the pi-plans workflow. Implement the accepted plan autonomously and emit [DONE:VC-xxx] markers in your replies as verifier items pass.";
811
- }
812
- }
813
-
814
- function delegatedExecutorTimeoutMs(): number | undefined {
815
- try {
816
- const stateRoot = resolveStateRootOrNull(process.cwd());
817
- if (stateRoot === null) return undefined;
818
- const minutes = loadConfig(stateRoot).executor_timeout_minutes;
819
- if (typeof minutes !== "number" || minutes <= 0) return undefined;
820
- return minutes * 60 * 1000;
821
- } catch {
822
- return undefined;
823
- }
824
661
  }
825
662
 
826
- /**
827
- * Delegated execution (R-9..R-12): spawn ONE executor child for the whole
828
- * plan (write-capable tools, chosen model, PI_PLANS_EXECUTOR=1 + pinned run
829
- * id), stream its progress into the overlay, parse full-text assistant
830
- * messages for [DONE:VC-xxx]/[I-xxx] markers, and on exit verify the
831
- * remaining items. Abort/timeout → stopExecution (stopped, resumable);
832
- * clean exit with items left → stay executing (resumable under either
833
- * runtime); clean exit complete → normal completion flow.
834
- */
835
- async function runDelegatedExecution(
836
- pi: ExtensionAPI,
837
- ctx: ExtensionContext,
838
- modelSelector: string,
839
- parentSignal: AbortSignal | undefined,
840
- ): Promise<void> {
663
+ /** Persist the live task progress (called by the task status tool). */
664
+ export function persistTaskProgress(ctx: ExtensionContext): void {
841
665
  if (!execution) return;
842
- const controller = new AbortController();
843
- delegatedRuntime = controller;
844
- const relayAbort = () => controller.abort();
845
- if (parentSignal?.aborted) controller.abort();
846
- else parentSignal?.addEventListener("abort", relayAbort, { once: true });
847
- const lang = resolveUiLanguage(ctx.cwd);
848
- const overlay = ctx.mode === "tui" ? new RefineOverlayController("executor", [{ id: "executor" }], relayAbort, lang) : undefined;
849
- if (overlay) {
850
- try {
851
- overlay.open(refineOverlayContext(ctx), modelSelector);
852
- } catch {
853
- /* overlay is best-effort; the blocking call itself must not fail */
854
- }
855
- }
856
- execution.delegate = { modelSelector, startedAt: utcNow() };
857
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: execution?.delegate ?? null }));
858
- persist(pi);
859
- const runId = executionRunId;
860
- const remainingAtStart = execution.items.filter((item) => !item.done).map((item) => item.id);
861
- const task = [
862
- `Implement the accepted plan at ${execution.planPath} (workdir: ${ctx.cwd}).`,
863
- "Read the plan file first; it is the source of truth for scope, sequencing, and verification steps.",
864
- remainingAtStart.length > 0
865
- ? `Verifier items still open: ${remainingAtStart.join(", ")}. Emit [DONE:VC-xxx] markers in your replies as each item's stated evidence passes.`
866
- : "All verifier items already passed; verify the plan end-to-end and report.",
867
- execution.implItems?.length
868
- ? `Implementation items: ${execution.implItems.map((item) => item.id).join(", ")} — emit [I-###:implemented]/[I-###:validating] markers as you progress.`
869
- : "",
870
- "Finish with the structured summary your system prompt specifies.",
871
- ]
872
- .filter((line) => line !== "")
873
- .join("\n");
874
- let result: Awaited<ReturnType<typeof runPiSubagent>>;
875
- try {
876
- result = await runPiSubagent({
877
- systemPrompt: executorAgentPrompt(),
878
- task,
879
- cwd: ctx.cwd,
880
- model: modelSelector,
881
- tools: ["read", "write", "edit", "bash", "grep", "find", "ls"],
882
- envMarker: "executor",
883
- runId: runId ?? undefined,
884
- signal: controller.signal,
885
- timeoutMs: delegatedExecutorTimeoutMs(),
886
- onProgress: (event: SubagentProgressEvent) => {
887
- try {
888
- overlay?.update("executor", event);
889
- } catch {
890
- /* display must not fail the child runner */
891
- }
892
- if (
893
- event.type === "transcript"
894
- && event.phase === "end"
895
- && event.entryType === "assistant-text"
896
- && typeof event.text === "string"
897
- ) {
898
- mirrorDelegateMarkers(pi, ctx, event.text);
899
- }
666
+ // Any task-state change resets the stall watchdog baseline.
667
+ execution.stall.rounds = 0;
668
+ execution.stall.lastSnapshot = stallSnapshot();
669
+ withExecutionCheckpoint(ctx, (cp) =>
670
+ applyExecutionProgress(cp, {
671
+ tasks: taskProgressMap(execution!.tasks),
672
+ doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
673
+ audit: {
674
+ rounds: execution!.audit.rounds,
675
+ lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
900
676
  },
901
- });
902
- } finally {
903
- delegatedRuntime = null;
904
- try {
905
- await overlay?.close();
906
- } catch {
907
- /* best-effort */
908
- }
909
- parentSignal?.removeEventListener("abort", relayAbort);
910
- }
911
- if (!result.ok) {
912
- const reason = result.timedOut
913
- ? `delegated executor timed out (${result.errorMessage ?? "no output"})`
914
- : result.cancelled || controller.signal.aborted
915
- ? "delegated executor aborted by user"
916
- : `delegated executor failed: ${result.errorMessage ?? "unknown error"}${result.stderr ? `; stderr: ${result.stderr.slice(0, 500)}` : ""}`;
917
- await stopExecution(pi, ctx, reason);
918
- return;
919
- }
920
- // Clean exit: land any markers from the final output text, then verify.
921
- mirrorDelegateMarkers(pi, ctx, result.output);
922
- if (isExecutionComplete()) {
923
- await completeExecution(pi, ctx);
924
- return;
925
- }
926
- // Items remain: keep the run executing (resumable via /plans-execute under
927
- // either runtime); the delegate bookkeeping is cleared so restarts do not
928
- // treat this as an orphaned child.
929
- if (execution) {
930
- execution.delegate = undefined;
931
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { delegate: null }));
932
- persist(pi);
933
- }
934
- const remaining = execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [];
935
- pi.sendMessage(
936
- {
937
- customType: "pi-plans-exec-delegate-exit",
938
- content: `**pi-plans: delegated executor exited with items remaining** — ${remaining.join(", ") || "(none)"}. Run stays executing; resume with /plans-execute (either runtime). Executor summary:\n${result.output.slice(0, 2000)}`,
939
- display: true,
940
- },
941
- { triggerTurn: false },
677
+ }),
942
678
  );
943
- ctx.ui.notify?.(`Delegated executor exited; ${remaining.length} verifier item(s) remain. Resume with /plans-execute.`, "warning");
944
- updateStatusWidget(ctx);
679
+ persist(ctx);
945
680
  }
946
681
 
947
- /** Apply VC/I markers parsed from a delegated child's full-text message. Exported for tests. */
948
- export function mirrorDelegateMarkers(pi: ExtensionAPI, ctx: ExtensionContext, text: string): void {
949
- const changedVc = applyDoneMarkers(text);
950
- const changedImpl = applyImplMarkers(text);
951
- applyCurrentIMarker(text);
952
- if (changedVc.length === 0 && changedImpl.length === 0) return;
953
- withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp));
954
- persist(pi);
955
- updateStatusWidget(ctx);
956
- }
957
-
958
- /** Record one assistant turn: accumulate usage and mark any completed items. */
682
+ /** Record one assistant turn: accumulate usage only (markers are gone). */
959
683
  export function recordExecutionTurn(
960
- pi: ExtensionAPI,
961
- _ctx: ExtensionContext,
962
- _completedIds: string[],
684
+ ctx: ExtensionContext,
963
685
  usage?: { input: number; output: number },
964
686
  ): void {
965
687
  if (!execution) return;
@@ -967,102 +689,238 @@ export function recordExecutionTurn(
967
689
  execution.usage.inToks += usage.input;
968
690
  execution.usage.outToks += usage.output;
969
691
  }
970
- // I-005: mirror progress into the run checkpoint so a different session
971
- // can resume with the verified VC/I set (R-004).
972
- withExecutionCheckpoint(_ctx, (cp) =>
692
+ withExecutionCheckpoint(ctx, (cp) =>
973
693
  applyExecutionProgress(cp, {
974
- doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
975
- implStatus: implStatusSnapshot(),
976
- currentI: execution!.currentI,
977
694
  usage: usage ? { inToks: usage.input, outToks: usage.output } : undefined,
978
695
  }),
979
696
  );
980
- requestExecutionFlush(pi, _ctx);
981
- updateStatusWidget(_ctx);
697
+ requestExecutionFlush();
698
+ updateStatusWidget(ctx);
982
699
  }
983
700
 
984
- function implStatusSnapshot(): Record<string, string> {
985
- const snapshot: Record<string, string> = {};
986
- if (!execution?.implItems) return snapshot;
987
- for (const item of execution.implItems) {
988
- const state = execution.implStatus?.[item.id];
989
- if (state) snapshot[item.id] = state;
990
- }
991
- return snapshot;
701
+ /** Test hook: replace the audit subagent with a deterministic function. */
702
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
703
+
704
+ export function __setAuditRunnerForTests(
705
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
706
+ ): void {
707
+ auditRunnerForTests = runner;
992
708
  }
993
709
 
994
710
  export function registerExecutionTurnHandlers(
995
- pi: ExtensionAPI,
711
+ ext: ExtensionAPI,
996
712
  onTurnEnd?: (ctx: ExtensionContext) => Promise<void> | void,
997
713
  ): void {
998
- // The turn_end projection does not carry usage; message_end delivers the
999
- // full assistant message, so cache it here and consume it per turn.
1000
714
  let lastAssistantUsage: { input: number; output: number } | null = null;
1001
- pi.on("agent_start", async (_event, ctx) => {
1002
- const runtime = currentGoalWaitRuntime(ctx);
715
+ ext.on("agent_start", async (_event, ctx) => {
716
+ const runtime = currentContinuationRuntime(ctx);
1003
717
  if (!runtime) return;
1004
718
  runtime.handled = false;
1005
719
  runtime.stopReason = undefined;
1006
- runtime.text = "";
720
+ // v0.7.1: a new agent run opens a new settle window (per-settle latch).
721
+ if (execution) resetSettleLatch();
1007
722
  });
1008
- pi.on("before_agent_start", async (_event, ctx) => {
1009
- const runtime = currentGoalWaitRuntime(ctx);
723
+ ext.on("before_agent_start", async (_event, ctx) => {
724
+ const runtime = currentContinuationRuntime(ctx);
1010
725
  if (runtime) runtime.wakeId = undefined;
1011
726
  });
1012
- pi.on("input", async (event, ctx) => {
1013
- if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(pi, ctx);
727
+ ext.on("input", async (event, ctx) => {
728
+ if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(ctx);
729
+ });
730
+ ext.on("agent_settled", async (_event, ctx) => {
731
+ drainExecutionFlush(ctx);
732
+ maybeContinuationFollowUp(ctx);
1014
733
  });
1015
- pi.on("agent_settled", async (_event, ctx) => {
1016
- drainExecutionFlush(pi, ctx);
1017
- maybeGoalWaitFollowUp(pi, ctx);
734
+ // v0.7.1: `agent_settled` is notification-only per pi's contract, so the
735
+ // audit fallback lives on `agent_before_settle` — the final ACTIONABLE
736
+ // boundary. It is what makes a terminal-but-unaudited run self-heal with
737
+ // zero user input, instead of stranding until a manual /plans-execute.
738
+ ext.on("agent_before_settle", async (_event, ctx) => {
739
+ if (!pendingAudit()) return;
740
+ const runtime = currentContinuationRuntime(ctx);
741
+ if (runtime?.handled) return;
742
+ // The continuation wake owns this settle: if the loop already woke the
743
+ // agent to fix rolled-back work, the audit waits for the next settle
744
+ // (Q-2 single-wake guarantee) rather than emitting a second wake.
745
+ if (runtime && !allTasksTerminal(runtime.owner.tasks)) {
746
+ maybeContinuationFollowUp(ctx);
747
+ return;
748
+ }
749
+ // Fully settled and still owed an audit: run it now. The audit's own
750
+ // pass/fail message drives the rest (pass completes the run; fail
751
+ // triggers a fix turn), so no extra continuation is requested here —
752
+ // pendingAudit() is false once paused, which bounds any loop.
753
+ latchAuditThisSettle();
754
+ await runAuditFlow(ctx);
1018
755
  });
1019
- pi.on("session_shutdown", async (_event, ctx) => {
1020
- drainExecutionFlush(pi, ctx);
756
+ ext.on("session_shutdown", async (_event, ctx) => {
757
+ drainExecutionFlush(ctx);
1021
758
  execution = null;
1022
759
  executionRunId = null;
1023
- goalWaitRuntime = null;
760
+ continuationRuntime = null;
1024
761
  lastAssistantUsage = null;
1025
762
  });
1026
- pi.on("message_end", async (event) => {
763
+ ext.on("message_end", async (event) => {
1027
764
  const message = event.message as { role?: string; usage?: { input?: number; output?: number } };
1028
765
  if (message?.role === "assistant" && message.usage) {
1029
766
  lastAssistantUsage = { input: message.usage.input ?? 0, output: message.usage.output ?? 0 };
1030
767
  }
1031
768
  });
769
+ // v0.7.1 (root cause B): the stall watchdog counted only task-status changes,
770
+ // so a round where the agent legitimately investigated (read code, gathered
771
+ // evidence) without closing a task looked identical to a dead agent. A
772
+ // SUCCESSFUL tool result is real progress; a failed/blocked tool is not, so
773
+ // an agent looping on the same error still trips the cap (F-005, Q-3).
774
+ ext.on("tool_result", async (event, _ctx) => {
775
+ if (!execution) return;
776
+ const result = event as { isError?: boolean; error?: unknown };
777
+ if (result.isError === true || result.error !== undefined) return;
778
+ auditLatchOf(execution).activity += 1; // Real progress: rebase the watchdog so this round counts as a change.
779
+ execution.stall.rounds = 0;
780
+ execution.stall.lastSnapshot = stallSnapshot();
781
+ });
1032
782
 
1033
- pi.on("turn_end", async (event, ctx) => {
1034
- const message = event.message as { role?: string; stopReason?: string; content?: Array<{ type: string; text?: string }> };
783
+ ext.on("turn_end", async (event, ctx) => {
784
+ const message = event.message as { role?: string; stopReason?: string };
1035
785
  if (!message || message.role !== "assistant") {
1036
786
  updateStatusWidget(ctx);
1037
787
  return;
1038
788
  }
1039
- const text = (message.content ?? [])
1040
- .filter((part) => part.type === "text")
1041
- .map((part) => part.text ?? "")
1042
- .join("\n");
1043
- const runtime = currentGoalWaitRuntime(ctx);
1044
- if (runtime) {
1045
- runtime.stopReason = message.stopReason;
1046
- runtime.text = text;
1047
- }
1048
- const changedIds = applyDoneMarkers(text);
1049
- const changedImpls = applyImplMarkers(text);
1050
- const changedCurrentI = applyCurrentIMarker(text);
789
+ const runtime = currentContinuationRuntime(ctx);
790
+ if (runtime) runtime.stopReason = message.stopReason;
1051
791
  const projection = (event.message as { usage?: { input?: number; output?: number } }).usage;
1052
792
  const raw = projection ?? lastAssistantUsage;
1053
- lastAssistantUsage = null; // consumed: never re-attribute a stale turn
793
+ lastAssistantUsage = null;
1054
794
  const usage = raw ? { input: raw.input ?? 0, output: raw.output ?? 0 } : undefined;
1055
- if (usage || changedIds.length > 0 || changedImpls.length > 0 || changedCurrentI) {
1056
- // Attribute this turn's usage now; `[DONE]` markers still only mark completion.
1057
- recordExecutionTurn(pi, ctx, changedIds, usage);
1058
- }
1059
- if (getExecution() && isExecutionComplete()) {
1060
- await completeExecution(pi, ctx);
795
+ if (usage) recordExecutionTurn(ctx, usage);
796
+ if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
797
+ latchAuditThisSettle();
798
+ await runAuditFlow(ctx);
1061
799
  }
1062
800
  await onTurnEnd?.(ctx);
1063
801
  });
1064
802
  }
1065
803
 
804
+ /** Audit flow: run the completion auditor, apply pass/rollback, and either
805
+ * complete the run, keep iterating (rollback), pause at the round cap
806
+ * (interactive), or stop at the cap (auto-approve/headless — D-022).
807
+ *
808
+ * Fail-closed completion: a run completes only when EVERY pending check was
809
+ * affirmatively passed (or resolved skipped-pass); checks the auditor failed
810
+ * to report count as failed, never as silently passed. */
811
+ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
812
+ if (!execution) return;
813
+ const ex = execution;
814
+ // Skipped-pass checks resolve without a subagent round.
815
+ for (const id of presolvedCheckIds(ex.items, ex.tasks)) {
816
+ const item = ex.items.find((candidate) => candidate.id === id);
817
+ if (item) item.done = true;
818
+ }
819
+ // Only auditable checks (with task coverage) gate completion; checks that
820
+ // cover no task can never be verified and never block or complete.
821
+ const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
822
+ if (pendingChecks.length === 0) {
823
+ await completeExecution(ctx);
824
+ return;
825
+ }
826
+ if (ex.audit.rounds >= AUDIT_MAX_ROUNDS) {
827
+ // D-022: interactive sessions pause for the user (state kept, tasks
828
+ // intact, resumable); auto-approve/headless terminates bounded.
829
+ if (isInteractiveSession(ctx)) {
830
+ pauseForStall(
831
+ ctx,
832
+ `${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
833
+ );
834
+ return;
835
+ }
836
+ await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
837
+ return;
838
+ }
839
+ ex.audit.running = true;
840
+ ex.audit.rounds += 1;
841
+ updateStatusWidget(ctx);
842
+ let outcome = null as Awaited<ReturnType<typeof runCompletionAudit>>;
843
+ try {
844
+ outcome = auditRunnerForTests
845
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: ex.audit.rounds })
846
+ : await runCompletionAudit(ctx, {
847
+ planPath: ex.planPath,
848
+ checklist: ex.items,
849
+ tasks: ex.tasks,
850
+ round: ex.audit.rounds,
851
+ signal: ctx.signal,
852
+ });
853
+ } finally {
854
+ ex.audit.running = false;
855
+ }
856
+ // Fail-closed: checks the outcome did not affirmatively pass are failed.
857
+ const reportedPass = new Set(outcome?.passed ?? []);
858
+ const failed = pendingChecks
859
+ .map((item) => item.id)
860
+ .filter((id) => !reportedPass.has(id));
861
+ if (failed.length === 0) {
862
+ ex.audit.failed = [];
863
+ withExecutionCheckpoint(ctx, (cp) =>
864
+ applyExecutionProgress(cp, {
865
+ tasks: taskProgressMap(ex.tasks),
866
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
867
+ audit: { rounds: ex.audit.rounds, passed: true },
868
+ }),
869
+ );
870
+ await completeExecution(ctx);
871
+ return;
872
+ }
873
+ // Failed checks: roll their covered tasks back (always — an unreported or
874
+ // infra-failed audit must reopen work so the loop can continue) and
875
+ // persist progress including checks that passed earlier rounds.
876
+ for (const id of failed) {
877
+ const item = ex.items.find((candidate) => candidate.id === id);
878
+ if (item) item.done = false;
879
+ }
880
+ ex.audit.failed = failed;
881
+ const rolledBack: string[] = [];
882
+ for (const id of failed) {
883
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
884
+ }
885
+ withExecutionCheckpoint(ctx, (cp) =>
886
+ applyExecutionProgress(cp, {
887
+ tasks: taskProgressMap(ex.tasks),
888
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
889
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
890
+ }),
891
+ );
892
+ ex.stall.rounds = 0;
893
+ ex.stall.lastSnapshot = stallSnapshot();
894
+ persist(ctx);
895
+ updateStatusWidget(ctx);
896
+ const report = outcome?.report ?? "(audit subagent failed to run)";
897
+ // v0.7.1: the failure notification now wakes the agent so it can fix the
898
+ // rolled-back work without the user having to poke the run (root cause of
899
+ // the observed stall). The wake is the ONLY turn this settle emits — the
900
+ // rollback below also marks the runtime handled so the continuation path
901
+ // cannot add a second EXECUTION_CONTINUE wake for the same settle (Q-2).
902
+ const stranded = rolledBack.length === 0;
903
+ messaging().sendMessage(
904
+ {
905
+ customType: "pi-plans-audit-failed",
906
+ content: `**pi-plans: completion audit round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the audit reruns automatically once all tasks are terminal again.${ex.audit.rounds >= AUDIT_MAX_ROUNDS ? ` This was round ${AUDIT_MAX_ROUNDS} of ${AUDIT_MAX_ROUNDS}: interactive sessions pause for review; the next terminal-task cycle stops or pauses the run.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the audit automatically.` : ""}\n\n---\n${report.slice(0, 4000)}`,
907
+ display: true,
908
+ },
909
+ { triggerTurn: true },
910
+ );
911
+ // Q-2 (single wake): this message owns the settle's only turn. Mark the
912
+ // runtime handled so maybeContinuationFollowUp stays silent.
913
+ const runtime = currentContinuationRuntime(ctx);
914
+ if (runtime) runtime.handled = true;
915
+ }
916
+
917
+ /** True when the session can surface a pause to a human (D-022): interactive
918
+ * TUI/RPC sessions that are not running under PI_PLANS_AUTO_APPROVE. */
919
+ function isInteractiveSession(ctx: ExtensionContext): boolean {
920
+ if ((ctx.mode !== "tui" && ctx.mode !== "rpc") || ctx.hasUI !== true) return false;
921
+ return !isAutoApproveEnabledLocal();
922
+ }
923
+
1066
924
  const EXECUTION_RESUME_CUSTOM_TYPE = "pi-plans-exec-resume";
1067
925
 
1068
926
  function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase): { settings: PiPlansVccSettings; runId: string; artifactDir: string } | null {
@@ -1079,12 +937,13 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
1079
937
  }
1080
938
 
1081
939
  function executionVccContext(): PiPlansVccPhaseContext {
940
+ const current = execution ? currentTask(execution.tasks) : null;
1082
941
  return {
1083
942
  phase: "execution",
1084
943
  planPath: execution?.planPath ?? null,
1085
- currentI: execution?.currentI ?? null,
944
+ currentI: current?.id ?? null,
1086
945
  remainingVerifierIds: execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [],
1087
- implementationIds: execution?.implItems?.map((item) => item.id) ?? [],
946
+ implementationIds: execution ? flattenTaskViews(execution.tasks).map((task) => task.id) : [],
1088
947
  };
1089
948
  }
1090
949
 
@@ -1128,7 +987,6 @@ export function buildExecutionCompactionResult(event: SessionBeforeCompactEvent,
1128
987
  }
1129
988
 
1130
989
  export function handleExecutionBeforeCompact(
1131
- pi: ExtensionAPI,
1132
990
  ctx: ExtensionContext,
1133
991
  event: SessionBeforeCompactEvent,
1134
992
  ): SessionBeforeCompactResult | undefined {
@@ -1159,7 +1017,7 @@ export function handleExecutionBeforeCompact(
1159
1017
  state.pendingStats = built.stats;
1160
1018
  state.pendingFollowUpPrompt = built.followUpPrompt;
1161
1019
  state.pendingContinueAfterThresholdCompact = built.settings.continueAfterThresholdCompact;
1162
- requestExecutionFlush(pi, ctx);
1020
+ requestExecutionFlush();
1163
1021
  return { compaction: built.compaction };
1164
1022
  }
1165
1023
 
@@ -1167,7 +1025,7 @@ function runtimePiVersion(ctx: ExtensionContext): unknown {
1167
1025
  return (ctx as ExtensionContext & { piVersion?: unknown }).piVersion ?? VERSION;
1168
1026
  }
1169
1027
 
1170
- export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1028
+ export async function handleExecutionCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1171
1029
  if (!execution) return;
1172
1030
  const state = ensureExecutionCompactionState(ctx);
1173
1031
  const stats = state.pendingStats;
@@ -1187,10 +1045,10 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
1187
1045
  if (!event.willRetry && stats) {
1188
1046
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
1189
1047
  if (followUpPrompt) {
1190
- await pi.sendUserMessage?.(followUpPrompt);
1048
+ await messaging().sendUserMessage(followUpPrompt);
1191
1049
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
1192
1050
  state.resumeGuard = true;
1193
- pi.sendMessage(
1051
+ messaging().sendMessage(
1194
1052
  {
1195
1053
  customType: EXECUTION_RESUME_CUSTOM_TYPE,
1196
1054
  content: EXECUTION_COMPACTION_RESUME_MESSAGE,
@@ -1200,18 +1058,15 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
1200
1058
  );
1201
1059
  }
1202
1060
  }
1203
- requestExecutionFlush(pi, ctx);
1061
+ requestExecutionFlush();
1204
1062
  updateStatusWidget(ctx);
1205
1063
  }
1206
1064
 
1207
-
1208
- export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1065
+ export function handleExecutionCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1209
1066
  if (!execution) return;
1210
1067
  const state = executionCompactionState(ctx);
1211
1068
  const terminal = isTerminalCompactionFailure(event);
1212
1069
  if (terminal) {
1213
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
1214
- // only after real growth or high-watermark pressure so the loop stops.
1215
1070
  if (state) {
1216
1071
  state.inFlight = false;
1217
1072
  state.resumeGuard = false;
@@ -1228,7 +1083,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
1228
1083
  ? "pi-plans: compaction found nothing to summarize; backing off until the session grows past the keep-recent window."
1229
1084
  : "pi-plans: compaction was aborted (provider interruption, user cancel, or a competing manual compact); backing off until the session grows or usage nears the window.";
1230
1085
  ctx.ui.notify(message, "info");
1231
- requestExecutionFlush(pi, ctx);
1086
+ requestExecutionFlush();
1232
1087
  return;
1233
1088
  }
1234
1089
  if (state) {
@@ -1245,7 +1100,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
1245
1100
  `pi-plans: compaction failed (${event.reason}); execution remains active and will wait for the next eligible turn.`,
1246
1101
  "warning",
1247
1102
  );
1248
- requestExecutionFlush(pi, ctx);
1103
+ requestExecutionFlush();
1249
1104
  }
1250
1105
 
1251
1106
  export function filterExecutionResumeMessages<T extends { customType?: string }>(messages: T[]): T[] {
@@ -1255,24 +1110,12 @@ export function filterExecutionResumeMessages<T extends { customType?: string }>
1255
1110
  // ---------------------------------------------------------------------------
1256
1111
  // Planning-phase compaction: Pi core owns scheduling; this hook customizes
1257
1112
  // active planning compact events with the same VCC builder used by execution.
1258
- // The two state machines are kept independent (different memory slot and
1259
- // snapshot key) so execution never bleeds into planning.
1260
1113
  // ---------------------------------------------------------------------------
1261
1114
 
1262
1115
  export const PLANNING_RUN_START_CUSTOM_TYPE = "pi-plans-run-start";
1263
1116
  export const PLANNING_PLAN_WRITTEN_CUSTOM_TYPE = "pi-plans-plan-written";
1264
1117
  const PLANNING_RESUME_CUSTOM_TYPE = "pi-plans-plan-resume";
1265
1118
 
1266
- // ---------------------------------------------------------------------------
1267
- // Pre-plan compaction: right after `plans start-run` creates a new planning
1268
- // run, the extension triggers one VCC compaction so the new plan starts on a
1269
- // lean context (LLM reasoning degrades with longer context; see PLAN
1270
- // preplan-compact). The pending flag is session-scoped and opportunistic: it
1271
- // is set by the start-run tool case and consumed by the plans tool_result
1272
- // hook in index.ts, which requests the extension-context compact action and
1273
- // resumes planning exactly once regardless of success or failure.
1274
- // ---------------------------------------------------------------------------
1275
-
1276
1119
  export { PLANNING_PREPLAN_COMPACT_HINT };
1277
1120
  export const PLANNING_PREPLAN_RESUME_CUSTOM_TYPE = "pi-plans-preplan-resume";
1278
1121
 
@@ -1292,11 +1135,8 @@ export function consumePrePlanCompactPending(ctx: ExtensionContext): PrePlanComp
1292
1135
  return pending;
1293
1136
  }
1294
1137
 
1295
- /** Hidden resume message after the pre-plan compaction settles (success or
1296
- * failure): Pi's manual compaction never continues the aborted turn, so the
1297
- * planning workflow is continued exactly once from here. */
1298
- export function sendPrePlanCompactResume(pi: ExtensionAPI): void {
1299
- pi.sendMessage?.(
1138
+ export function sendPrePlanCompactResume(ctx: ExtensionContext): void {
1139
+ messaging().sendMessage(
1300
1140
  {
1301
1141
  customType: PLANNING_PREPLAN_RESUME_CUSTOM_TYPE,
1302
1142
  content: "Continue planning.",
@@ -1313,7 +1153,6 @@ interface PlanningCompactionState {
1313
1153
  lastAttemptReason: "manual" | "threshold" | "overflow" | null;
1314
1154
  lastSuccessfulUsagePercent: number | null;
1315
1155
  lastSuccessfulAt: string | null;
1316
- /** Terminal "nothing to compact" backoff: tokens observed when Pi refused. */
1317
1156
  terminalBackoffTokens: number | null;
1318
1157
  pendingStats: VccCompactionStats | null;
1319
1158
  pendingFollowUpPrompt: string | null;
@@ -1341,8 +1180,6 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1341
1180
  if (message.includes("nothing to compact") || message.includes("already compacted") || message.includes("session too small")) {
1342
1181
  return { kind: "content" };
1343
1182
  }
1344
- // abort/stream class: explicit event names only, so that provider blips
1345
- // (network down, etc.) stay retryable.
1346
1183
  const abortPatterns = [
1347
1184
  "this operation was aborted",
1348
1185
  "aborted",
@@ -1354,20 +1191,12 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1354
1191
  if (abortPatterns.some((pattern) => message.includes(pattern))) {
1355
1192
  return { kind: "abort-stream" };
1356
1193
  }
1357
- // Aborted with no recognized message: still an abort-class terminal so the
1358
- // next eligible turn does not immediately retry the same operation.
1359
1194
  if (event.aborted === true) {
1360
1195
  return { kind: "abort-stream" };
1361
1196
  }
1362
1197
  return null;
1363
1198
  }
1364
1199
 
1365
- /** Session-scoped phase-local "compaction in flight" guard.
1366
- * - Set on `session_before_compact` for the phase attributed by the custom
1367
- * instructions hint; auto-compaction (no hint) marks both phases defensively.
1368
- * - Cleared on `session_compact` and `session_compact_failed`.
1369
- * - Retained so lifecycle events expose the same phase-local state to tests
1370
- * and future Pi core schema additions. */
1371
1200
  type CompactionPhase = "planning" | "execution";
1372
1201
 
1373
1202
  function compactionLifecycleStore(ctx: ExtensionContext): {
@@ -1396,18 +1225,11 @@ export function noteCompactionStarted(ctx: ExtensionContext, customInstructions:
1396
1225
  } else if (isExecutionCustomInstructions(customInstructions)) {
1397
1226
  store.execution = true;
1398
1227
  } else {
1399
- // Auto-compaction (threshold/overflow/manual without our hint) marks both.
1400
1228
  store.planning = true;
1401
1229
  store.execution = true;
1402
1230
  }
1403
1231
  }
1404
1232
 
1405
- /** Pi core's `SessionCompactEvent` / `SessionCompactFailedEvent` do not carry
1406
- * `customInstructions` in any emission site, so the END side has no way to
1407
- * know which phase the compaction belonged to. Clearing both phases is the
1408
- * safe default — the per-phase start side (above) already encodes the hint
1409
- * attribution. The hint parameter is retained for API symmetry and future
1410
- * Pi core schema additions. */
1411
1233
  export function noteCompactionEnded(ctx: ExtensionContext, _customInstructions: unknown): void {
1412
1234
  const store = compactionLifecycleStore(ctx);
1413
1235
  store.planning = false;
@@ -1431,15 +1253,11 @@ export function consumePlanningCompactionResumeGuard(ctx: ExtensionContext): boo
1431
1253
  }
1432
1254
 
1433
1255
  export function refreshPlanningCompactionCooldown(_ctx: ExtensionContext): void {
1434
- // Pi core owns scheduling; retained for lifecycle compatibility only.
1256
+ // Retained for lifecycle compatibility only.
1435
1257
  }
1436
1258
 
1437
1259
  export function requestPlanningCompaction(_ctx: ExtensionContext): void {
1438
- // Generic proactive pi-plans compaction is intentionally disabled. Manual,
1439
- // threshold, and overflow compactions are handled by session_before_compact.
1440
- // The single exception is the pre-plan compaction: index.ts requests the
1441
- // extension-context compact action from the plans tool_result hook right
1442
- // after start-run (see PLANNING_PREPLAN_COMPACT_HINT).
1260
+ // Manual, threshold, and overflow compactions are handled by session_before_compact.
1443
1261
  }
1444
1262
 
1445
1263
  function buildPlanningVccResult(event: SessionBeforeCompactEvent, ctx: ExtensionContext): VccCompactionBuildResult | null {
@@ -1464,7 +1282,6 @@ export function buildPlanningCompactionResult(event: SessionBeforeCompactEvent,
1464
1282
  }
1465
1283
 
1466
1284
  export function handlePlanningBeforeCompact(
1467
- pi: ExtensionAPI,
1468
1285
  ctx: ExtensionContext,
1469
1286
  event: SessionBeforeCompactEvent,
1470
1287
  ): SessionBeforeCompactResult | undefined {
@@ -1498,7 +1315,7 @@ export function handlePlanningBeforeCompact(
1498
1315
  return { compaction: built.compaction };
1499
1316
  }
1500
1317
 
1501
- export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1318
+ export async function handlePlanningCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1502
1319
  if (getExecution()) return;
1503
1320
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1504
1321
  const state = session.__planningCompaction;
@@ -1519,10 +1336,10 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1519
1336
  if (!event.willRetry && stats) {
1520
1337
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
1521
1338
  if (followUpPrompt) {
1522
- await pi.sendUserMessage?.(followUpPrompt);
1339
+ await messaging().sendUserMessage(followUpPrompt);
1523
1340
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
1524
1341
  state.resumeGuard = true;
1525
- pi.sendMessage(
1342
+ messaging().sendMessage(
1526
1343
  {
1527
1344
  customType: PLANNING_RESUME_CUSTOM_TYPE,
1528
1345
  content: "Continue planning.",
@@ -1534,16 +1351,13 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1534
1351
  }
1535
1352
  }
1536
1353
 
1537
-
1538
- export function handlePlanningCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1354
+ export function handlePlanningCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1539
1355
  if (getExecution()) return;
1540
1356
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1541
1357
  const state = session.__planningCompaction;
1542
1358
  if (!state) return;
1543
1359
  const terminal = isTerminalCompactionFailure(event);
1544
1360
  if (terminal) {
1545
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
1546
- // only after real growth or high-watermark pressure so the loop stops.
1547
1361
  state.inFlight = false;
1548
1362
  state.resumeGuard = false;
1549
1363
  state.cooldownActive = true;
@@ -1576,19 +1390,17 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
1576
1390
  return messages.filter((message) => message.customType !== PLANNING_RESUME_CUSTOM_TYPE && message.customType !== PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
1577
1391
  }
1578
1392
 
1579
- export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): Promise<void> {
1393
+ export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
1580
1394
  if (!execution) return;
1581
1395
  resetExecutionCompactionState(ctx);
1582
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1583
1396
  pendingExecutionFlush = false;
1584
- persist(pi);
1585
- // Checkpoint first: withExecutionCheckpoint guards on the live execution.
1397
+ persist(ctx);
1586
1398
  withExecutionCheckpoint(ctx, (cp) => applyExecutionStopped(cp, reason));
1587
1399
  execution = null;
1588
1400
  executionRunId = null;
1589
- goalWaitRuntime = null;
1590
- pi.appendEntry("pi-plans-exec-cleared", { reason });
1591
- pi.sendMessage(
1401
+ continuationRuntime = null;
1402
+ messaging().appendEntry("pi-plans-exec-cleared", { reason });
1403
+ messaging().sendMessage(
1592
1404
  {
1593
1405
  customType: "pi-plans-exec-stop",
1594
1406
  content: `**pi-plans: execution stopped** — ${reason}`,
@@ -1607,84 +1419,76 @@ export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, rea
1607
1419
  updateStatusWidget(ctx);
1608
1420
  }
1609
1421
 
1610
- /** Apply [DONE:VC-xxx] markers from an assistant message. Returns changed ids. */
1611
- export function applyDoneMarkers(text: string): string[] {
1612
- if (!execution) return [];
1613
- const changed: string[] = [];
1614
- for (const id of scanDoneMarkers(text)) {
1615
- const item = execution.items.find((candidate) => candidate.id === id && !candidate.done);
1616
- if (item) {
1617
- item.done = true;
1618
- changed.push(id);
1619
- }
1620
- }
1621
- return changed;
1422
+ /** v0.7.1: the per-settle latch, created on demand so no execution-construction
1423
+ * path can leave it undefined (a missing latch must degrade to "no latch",
1424
+ * never throw inside a lifecycle handler). */
1425
+ function auditLatchOf(ex: ExecState): ExecState["auditLatch"] {
1426
+ if (!ex.auditLatch) ex.auditLatch = { auditedThisSettle: false, activity: 0 };
1427
+ return ex.auditLatch;
1622
1428
  }
1623
1429
 
1624
- /**
1625
- * Apply [I-xxx:implemented|validating] markers from an assistant message.
1626
- * Unknown I-ids are silently ignored; later markers overwrite earlier ones.
1627
- * Returns the ids whose state actually changed.
1628
- */
1629
- export function applyImplMarkers(text: string): string[] {
1630
- if (!execution?.implItems?.length) return [];
1631
- const known = new Set(execution.implItems.map((impl) => impl.id));
1632
- execution.implStatus ??= {};
1633
- const changed: string[] = [];
1634
- for (const marker of scanImplMarkers(text)) {
1635
- if (!known.has(marker.id)) continue;
1636
- const previous = execution.implStatus[marker.id];
1637
- execution.implStatus[marker.id] = marker.state;
1638
- if (previous !== marker.state) changed.push(marker.id);
1639
- }
1640
- return changed;
1430
+ function stallSnapshot(): string {
1431
+ if (!execution) return "";
1432
+ // v0.7.1: the snapshot carries the round's tool-activity counter, so a round
1433
+ // in which the agent legitimately did work (read code, run commands) counts
1434
+ // as progress even when no task changed status. Only a round with neither a
1435
+ // status change NOR a successful tool result is "no progress".
1436
+ return JSON.stringify({ tasks: taskProgressMap(execution.tasks), activity: auditLatchOf(execution).activity });
1641
1437
  }
1642
1438
 
1643
- export function applyCurrentIMarker(text: string): boolean {
1644
- if (!execution?.implItems?.length) return false;
1645
- const markers = scanCurrentIMarkers(text);
1646
- const resolved = resolveCurrentI(execution.implItems, markers, execution.currentI);
1647
- if (!resolved || resolved === execution.currentI) return false;
1648
- execution.currentI = resolved;
1649
- return true;
1439
+ /**
1440
+ * v0.7.1: shared "the completion audit is owed" predicate. Every entry point
1441
+ * that can start the audit (turn_end, agent_before_settle, restoreFromSession,
1442
+ * the resume path) goes through this so they can never disagree.
1443
+ *
1444
+ * Semantics (unchanged from restoreFromSession's guard, F-006): a check that
1445
+ * covers no task never gates completion, so only auditable checks count.
1446
+ */
1447
+ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1448
+ if (!ex) return false;
1449
+ // A paused run is never self-driven: the stall / audit-cap pause is an
1450
+ // explicit "hand control back" signal, and resuming it is the user's call.
1451
+ // This also bounds the zero-input continue loop in agent_before_settle.
1452
+ if (ex.stall.paused) return false;
1453
+ if (ex.audit.running) return false;
1454
+ return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
1650
1455
  }
1651
1456
 
1652
- export function isExecutionComplete(): boolean {
1653
- return execution !== null && execution.items.length > 0 && execution.items.every((item) => item.done);
1457
+ /** v0.7.1: record that this settle already ran (or declined) its audit, so a
1458
+ * second entry point in the same settle cannot re-consume a round. */
1459
+ function latchAuditThisSettle(): void {
1460
+ if (execution) auditLatchOf(execution).auditedThisSettle = true;
1654
1461
  }
1655
1462
 
1656
- function goalWaitSnapshot(): string {
1657
- if (!execution) return "";
1658
- return JSON.stringify({
1659
- done: execution.items
1660
- .filter((item) => item.done)
1661
- .map((item) => item.id)
1662
- .sort()
1663
- .join("|"),
1664
- implStatus: execution.implStatus ?? {},
1665
- currentI: execution.currentI ?? null,
1666
- });
1463
+ /** v0.7.1: called on agent_start — a new agent run is a new settle window, so
1464
+ * the latch and the activity counter both reset here. */
1465
+ function resetSettleLatch(): void {
1466
+ if (!execution) return;
1467
+ const latch = auditLatchOf(execution);
1468
+ latch.auditedThisSettle = false;
1469
+ latch.activity = 0;
1667
1470
  }
1668
1471
 
1669
- function pauseGoalWait(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): void {
1472
+ function pauseForStall(ctx: ExtensionContext, reason: string): void {
1670
1473
  const ex = getExecution();
1671
- if (!ex?.goalWait) return;
1672
- ex.goalWait.paused = true;
1673
- ex.goalWait.pausedReason = reason;
1674
- persist(pi);
1474
+ if (!ex) return;
1475
+ ex.stall.paused = true;
1476
+ ex.stall.pausedReason = reason;
1477
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: reason }));
1478
+ persist(ctx);
1675
1479
  ctx.ui.notify?.(
1676
- `pi-plans: goal-wait paused (${reason}). Send any message or run /plans-execute to resume.`,
1480
+ `pi-plans: execution paused (${reason}). Send any message or run /plans-execute to resume.`,
1677
1481
  "warning",
1678
1482
  );
1679
1483
  updateStatusWidget(ctx);
1680
1484
  }
1681
1485
 
1682
- function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1486
+ function canWakeExecution(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1683
1487
  const compaction = executionCompactionState(ctx);
1684
- return currentGoalWaitRuntime(ctx) === runtime
1488
+ return currentContinuationRuntime(ctx) === runtime
1685
1489
  && (ctx.mode === "tui" || ctx.mode === "rpc")
1686
- && !isExecutionComplete()
1687
- && !runtime.owner.goalWait?.paused
1490
+ && !allTasksTerminal(runtime.owner.tasks)
1491
+ && !runtime.owner.stall.paused
1688
1492
  && ctx.isIdle()
1689
1493
  && !ctx.hasPendingMessages()
1690
1494
  && !ctx.signal?.aborted
@@ -1694,15 +1498,14 @@ function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): bool
1694
1498
  && compaction?.pendingFollowUpPrompt == null;
1695
1499
  }
1696
1500
 
1697
- function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1501
+ function sendContinuationWake(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1698
1502
  if (!canWakeExecution(ctx, runtime)) return false;
1699
1503
  try {
1700
- // Custom messages bypass before_agent_start, so carry fresh execution rules.
1701
1504
  const content = executionContextMessage(ctx);
1702
1505
  if (!content) return false;
1703
1506
  runtime.wakeId = randomUUID();
1704
- pi.sendMessage({
1705
- customType: GOAL_WAIT_CUSTOM_TYPE,
1507
+ messaging().sendMessage({
1508
+ customType: EXECUTION_CONTINUE_CUSTOM_TYPE,
1706
1509
  content,
1707
1510
  display: false,
1708
1511
  details: { wakeId: runtime.wakeId },
@@ -1710,120 +1513,133 @@ function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: Goal
1710
1513
  return true;
1711
1514
  } catch (error) {
1712
1515
  runtime.wakeId = undefined;
1713
- pauseGoalWait(pi, ctx, `continuation failed: ${String(error)}`);
1516
+ pauseForStall(ctx, `continuation failed: ${String(error)}`);
1714
1517
  return false;
1715
1518
  }
1716
1519
  }
1717
1520
 
1718
1521
  /** Only a fully settled agent run can need an extra wake, never a tool turn. */
1719
- function maybeGoalWaitFollowUp(pi: ExtensionAPI, ctx: ExtensionContext): void {
1720
- const runtime = currentGoalWaitRuntime(ctx);
1522
+ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
1523
+ const runtime = currentContinuationRuntime(ctx);
1721
1524
  if (!runtime || runtime.handled || !ctx.isIdle()) return;
1722
1525
  if (ctx.mode !== "tui" && ctx.mode !== "rpc") return;
1723
1526
  if (runtime.stopReason === "error" || runtime.stopReason === "aborted" || ctx.signal?.aborted) {
1724
1527
  runtime.handled = true;
1725
- pauseGoalWait(pi, ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1528
+ pauseForStall(ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1726
1529
  return;
1727
1530
  }
1531
+ // v0.7.1: a fully-terminal run that still owes an audit is NOT a
1532
+ // continuation case — the audit owns that state (agent_before_settle).
1533
+ // Return without waking: emitting EXECUTION_CONTINUE here would send the
1534
+ // agent back to redo work it has already finished (F-003).
1535
+ if (execution && allTasksTerminal(execution.tasks) && pendingAudit(execution)) return;
1728
1536
  if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
1729
1537
  runtime.handled = true;
1730
1538
  const ex = runtime.owner;
1731
- ex.goalWait ??= { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false };
1732
- const goalWait = ex.goalWait;
1733
- const snapshot = goalWaitSnapshot();
1734
- const changed = goalWait.lastMarkers !== null && snapshot !== goalWait.lastMarkers;
1735
- goalWait.lastMarkers = snapshot;
1539
+ const snapshot = stallSnapshot();
1540
+ const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
1541
+ ex.stall.lastSnapshot = snapshot;
1736
1542
  if (changed) {
1737
- goalWait.noProgressRounds = 0;
1738
- goalWait.waitRounds = 0;
1739
- } else if (/waiting for/i.test(runtime.text)) {
1740
- goalWait.waitRounds += 1;
1543
+ ex.stall.rounds = 0;
1741
1544
  } else {
1742
- goalWait.noProgressRounds += 1;
1743
- }
1744
- if (goalWait.noProgressRounds >= GOAL_WAIT_MAX_NO_PROGRESS) {
1745
- pauseGoalWait(pi, ctx, `no progress in ${goalWait.noProgressRounds} rounds`);
1746
- return;
1545
+ ex.stall.rounds += 1;
1747
1546
  }
1748
- if (goalWait.waitRounds >= GOAL_WAIT_MAX_WAITING) {
1749
- pauseGoalWait(pi, ctx, `waiting without progress for ${goalWait.waitRounds} rounds`);
1547
+ if (ex.stall.rounds >= STALL_MAX_ROUNDS) {
1548
+ pauseForStall(ctx, `no task-status change in ${ex.stall.rounds} rounds`);
1750
1549
  return;
1751
1550
  }
1752
- persist(pi);
1551
+ persist(ctx);
1753
1552
  updateStatusWidget(ctx);
1754
- // No await between the live gate and dispatch: another input cannot interleave.
1755
- sendGoalWaitWake(pi, ctx, runtime);
1553
+ sendContinuationWake(ctx, runtime);
1756
1554
  }
1757
1555
 
1758
1556
  export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
1759
- return messages.filter((message) => message.customType !== GOAL_WAIT_CUSTOM_TYPE
1760
- || (goalWaitRuntime?.owner === execution && goalWaitRuntime?.wakeId !== undefined
1761
- && (message.details as { wakeId?: unknown } | undefined)?.wakeId === goalWaitRuntime.wakeId));
1557
+ // v0.6.1: continuation wakes are one-shot; stale ones (including the
1558
+ // legacy v0.6.0 goal-wait type) never replay after a restart.
1559
+ return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
1560
+ && message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
1762
1561
  }
1763
1562
 
1764
- /** Called only for genuine user input or an explicit same-execution resume. */
1765
- export function resumeGoalWaitIfPaused(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
1563
+ export function filterContinuationMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
1564
+ return filterGoalWaitMessages(messages);
1565
+ }
1566
+
1567
+ /** Prefix of the stall reason used for the audit-cap pause (D-022). */
1568
+ const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
1569
+
1570
+ /** Called for genuine user input or an explicit same-execution resume.
1571
+ * Resuming an audit-cap pause grants a fresh audit budget (three more
1572
+ * rounds): the user's explicit resume IS the decision to keep auditing —
1573
+ * without this reset the cap pause could never be lifted productively. */
1574
+ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
1766
1575
  const ex = getExecution();
1767
- if (!ex?.goalWait?.paused || !currentGoalWaitRuntime(ctx)) return false;
1768
- ex.goalWait.paused = false;
1769
- ex.goalWait.pausedReason = undefined;
1770
- ex.goalWait.noProgressRounds = 0;
1771
- ex.goalWait.waitRounds = 0;
1772
- ex.goalWait.lastMarkers = goalWaitSnapshot();
1773
- persist(pi);
1576
+ if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
1577
+ const wasAuditCap = (ex.stall.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
1578
+ ex.stall.paused = false;
1579
+ ex.stall.pausedReason = undefined;
1580
+ ex.stall.rounds = 0;
1581
+ ex.stall.lastSnapshot = stallSnapshot();
1582
+ if (wasAuditCap) {
1583
+ ex.audit.rounds = 0;
1584
+ ex.audit.failed = [];
1585
+ withExecutionCheckpoint(ctx, (cp) =>
1586
+ applyExecutionProgress(cp, {
1587
+ tasks: taskProgressMap(ex.tasks),
1588
+ audit: { rounds: 0, lastResult: undefined },
1589
+ pausedReason: null,
1590
+ }),
1591
+ );
1592
+ } else {
1593
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
1594
+ }
1595
+ persist(ctx);
1774
1596
  updateStatusWidget(ctx);
1775
1597
  return true;
1776
1598
  }
1777
1599
 
1778
- export function resumeActiveExecution(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
1779
- if (!resumeGoalWaitIfPaused(pi, ctx)) return false;
1780
- const runtime = currentGoalWaitRuntime(ctx)!;
1600
+ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
1601
+ // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
1602
+ // to `return false` here, so `/plans-execute` answered "already executing"
1603
+ // and the run stayed stranded until a full re-entry or a session restore.
1604
+ // It is not paused, so the pause path below cannot see it — check it first
1605
+ // and run the owed audit instead of reporting "nothing to resume".
1606
+ if (!execution?.stall.paused && pendingAudit()) {
1607
+ latchAuditThisSettle();
1608
+ void runAuditFlow(ctx).catch(() => { /* surfaced via the audit message */ });
1609
+ return true;
1610
+ }
1611
+ if (!resumeGoalWaitIfPaused(ctx)) return false;
1612
+ const runtime = currentContinuationRuntime(ctx)!;
1781
1613
  if (canWakeExecution(ctx, runtime)) {
1782
1614
  runtime.handled = true;
1783
- sendGoalWaitWake(pi, ctx, runtime);
1615
+ sendContinuationWake(ctx, runtime);
1784
1616
  }
1785
1617
  return true;
1786
1618
  }
1787
1619
 
1788
- export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext): Promise<void> {
1620
+ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1789
1621
  if (!execution) return;
1790
1622
  resetExecutionCompactionState(ctx);
1791
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1792
1623
  pendingExecutionFlush = false;
1793
- persist(pi);
1794
-
1795
- const summary = execution.items.map((item) => `- ✅ \`${item.id}\` ${item.text.split(";")[0]}`).join("\n");
1624
+ persist(ctx);
1625
+ const flat = flattenTaskViews(execution.tasks);
1626
+ const summary = flat
1627
+ .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
1628
+ .join("\n");
1796
1629
  const planPath = execution.planPath;
1797
- // Checkpoint first (live-execution guard), then clear the session state.
1798
1630
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
1799
1631
  execution = null;
1800
1632
  executionRunId = null;
1801
- goalWaitRuntime = null;
1802
- pi.appendEntry("pi-plans-exec-cleared", { reason: "complete" });
1803
- // Post-execution goal-running continuation: in interactive sessions, attach
1804
- // the continuation block and trigger a new turn so the agent immediately
1805
- // enters the implementation-review loop. Headless sessions keep the silent
1806
- // completion behavior. Both completeExecution call sites (turn_end and the
1807
- // restoreFromSession recovery path) share this behavior.
1808
- const interactive = ctx.hasUI === true;
1809
- // Skill-aware continuation: the reviewer-count default follows the active
1810
- // run's skill (D-1/D-4), so the prompt names the run's own recommended
1811
- // count instead of a static guess.
1812
- const activeRunForPrompt = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1813
- const content = interactive
1814
- ? `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}\n\n${ameliorationPromptText(activeRunForPrompt?.skill)}`
1815
- : `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}`;
1816
- pi.sendMessage(
1633
+ continuationRuntime = null;
1634
+ messaging().appendEntry("pi-plans-exec-cleared", { reason: "complete" });
1635
+ messaging().sendMessage(
1817
1636
  {
1818
1637
  customType: "pi-plans-complete",
1819
- content,
1638
+ content: `**Plan complete!** ✅ \`${planPath}\` — completion audit passed.\n\n${summary}`,
1820
1639
  display: true,
1821
1640
  },
1822
- { triggerTurn: interactive },
1641
+ { triggerTurn: false },
1823
1642
  );
1824
- if (interactive) {
1825
- pi.appendEntry("pi-plans-ameliorate", { planPath, phase: "goal-started", rounds: null, currentRound: 0 });
1826
- }
1827
1643
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1828
1644
  if (active) {
1829
1645
  try {
@@ -1835,55 +1651,45 @@ export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext)
1835
1651
  updateStatusWidget(ctx);
1836
1652
  }
1837
1653
 
1838
- /** Instructions appended to the post-execution completion message in
1839
- * interactive sessions, telling the agent to enter the goal-running
1840
- * implementation-review loop. Skill-aware: the reviewer-count question's
1841
- * recommended option follows the run's skill (plan-big / plan-with-refs → 3,
1842
- * others → 1). Termination options are single-sourced from
1843
- * src/termination-prompt.ts (shared with the ask_choice trailing branch). */
1844
- export function ameliorationPromptText(skill: string | undefined): string {
1845
- return `---
1846
- Goal-running continuation: immediately ask the user now via ask_choice (autoComplete: false, in the session language) the termination question: "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(skill)} Then keep running the implementation-review loop without asking whether to continue; the goal-wait option keeps the loop running until no unpassed VCs remain.`;
1847
- }
1848
-
1849
1654
  /** Injection text for before_agent_start while executing. */
1850
1655
  export function executionContextMessage(ctx: ExtensionContext): string | null {
1851
1656
  if (!execution) return null;
1852
- const remaining = execution.items.filter((item) => !item.done);
1853
- const list =
1854
- remaining.map((item) => `- \`${item.id}\` ${item.text}`).join("\n") || "(none — report completion now)";
1855
- // Live read: the injected guidance and the tool wrappers share the same
1856
- // tri-state, so they can never contradict each other mid-run.
1657
+ const flat = flattenTaskViews(execution.tasks);
1658
+ const open = flat.filter((task) => !taskIsTerminal(task));
1659
+ const cur = currentTask(execution.tasks);
1660
+ const currentWave = cur?.wave ?? 1;
1661
+ const inWave = open.filter((task) => task.wave === currentWave);
1662
+ const waveList = inWave.map((task) => `- ${task.id}${task.children.length ? ` (${task.children.map((c) => c.id).join(", ")})` : ""}: ${task.title}${task.files.length ? ` — files: ${task.files.join(", ")}` : ""}`).join("\n") || "(none — take the next wave)";
1663
+ const progress = taskProgress(execution.tasks);
1664
+ const vcDone = execution.items.filter((item) => item.done).length;
1857
1665
  const mode = resolveGraphMode(ctx?.cwd ?? process.cwd());
1858
1666
  const graphLine =
1859
1667
  mode === "config-unavailable"
1860
- ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi_plans/config.json is repaired]`
1668
+ ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
1861
1669
  : graphBlockForExecutor(mode === "enabled");
1862
- // F-002 (impl review r1): the next-action line is hoisted out of the
1863
- // implItems ternary so plans without implementation items get the same
1864
- // same-source guidance the panel shows.
1865
- const nextActionLine = `\nSuggested next action (displayed in the pi-plans panel): ${deriveNextAction(execution, executionIsWaiting(execution), resolveImplStatuses(execution.implItems ?? [], execution.items, execution.implStatus), remaining, execution.currentI)}`;
1866
- const implementationItems = execution.implItems?.length
1867
- ? `\nImplementation items: ${execution.implItems.map((item) => item.id).join(", ")}${execution.currentI ? `\nCurrent implementation item: \`${execution.currentI}\`` : ""}\nWhen beginning an implementation item, emit its current anchor exactly once as \`[I-###:current]\`; then use \`[I-###:implemented]\` or \`[I-###:validating]\` for progress.`
1670
+ const rollbackNote = execution.audit.failed.length > 0
1671
+ ? `\nCompletion audit round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
1868
1672
  : "";
1869
1673
  return `[PI-PLANS EXECUTION — write access enabled]
1870
- Implement the accepted plan at ${execution.planPath} (${execution.items.length - remaining.length}/${execution.items.length} verifier items done).
1674
+ Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
1871
1675
 
1872
- Remaining verifier items:
1873
- ${list}${implementationItems}${nextActionLine}
1676
+ Current wave ${currentWave} open tasks:
1677
+ ${waveList}
1678
+
1679
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
1874
1680
 
1875
1681
  ${graphLine}
1876
1682
 
1877
1683
  Execution rules:
1878
- - Implement implementation items in dependency order; grow the change in layers — smallest end-to-end slice first, then stack each new capability on top of what already works.
1879
- - Report implementation-item progress with lightweight markers in your reply: write \`[I-001:implemented]\` when an item's code is done, \`[I-001:validating]\` when you start verifying it. The execution status bar tracks these states.
1880
- - For subprocess-backed verification, when a step starts a subprocess and needs its result before verifying, use literal \`waiting for\` with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s; restart at 5s for each new subprocess.
1881
- - Simplest implementation that fully meets the item: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1882
- - Architectural decisions are for the long term: no stopgaps. Do not add backward-compatibility layers, fallbacks, or migrations — remove the obsolete paths this change obsoletes.
1883
- - Prefer established, well-maintained libraries when they reduce complexity or improve reliability; before writing your own implementation or adding a package, check the project's existing dependencies (docs and types) — never reimplement common functionality without a clear reason.
1884
- - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists; when unsure, skip and emit \`[test skipped: <name>, add when <trigger>]\`.
1885
- - After verifying an item's pass condition with its stated evidence, include \`[DONE:VC-xxx]\` in your reply.
1886
- - When every item is done, report a completion summary.`;
1684
+ - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
1685
+ - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
1686
+ - Close subtasks before their parent; a parent is auditable only when every child is terminal.
1687
+ - When every task is terminal, the independent completion auditor verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
1688
+ - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1689
+ - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
1690
+ - Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
1691
+ - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists.
1692
+ - For subprocess-backed verification, when a step needs a subprocess result before proceeding, poll with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s.`;
1887
1693
  }
1888
1694
 
1889
1695
  interface SessionEntry {
@@ -1895,24 +1701,21 @@ interface SessionEntry {
1895
1701
 
1896
1702
  /**
1897
1703
  * Rebuild execution state from the session on start/resume. Finds the last
1898
- * pi-plans-exec snapshot, then re-scans assistant messages after it for
1899
- * [DONE:VC-xxx] markers so progress survives restarts.
1704
+ * pi-plans-exec snapshot; the task tree is rebuilt from the persisted
1705
+ * snapshot (tool-driven progress survives restarts without text replay).
1900
1706
  */
1901
- export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
1902
- pendingExecutionFlush = false; // no flush debt survives a restart
1903
- goalWaitRuntime = null;
1707
+ export async function restoreFromSession(ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
1708
+ pendingExecutionFlush = false;
1709
+ continuationRuntime = null;
1904
1710
  resetExecutionCompactionState(ctx);
1905
- let snapshotIndex = -1;
1906
1711
  let snapshot: ExecState | null = null;
1907
1712
  for (let i = entries.length - 1; i >= 0; i--) {
1908
1713
  const entry = entries[i];
1909
1714
  if (entry.type === "custom" && entry.customType === "pi-plans-exec" && entry.data) {
1910
1715
  snapshot = entry.data;
1911
- snapshotIndex = i;
1912
1716
  break;
1913
1717
  }
1914
1718
  if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1915
- // Execution was explicitly stopped or completed after the last snapshot.
1916
1719
  execution = null;
1917
1720
  updateStatusWidget(ctx);
1918
1721
  return;
@@ -1923,78 +1726,48 @@ export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext
1923
1726
  updateStatusWidget(ctx);
1924
1727
  return;
1925
1728
  }
1926
- // Ignore stale plans whose file vanished.
1927
1729
  if (!fs.existsSync(snapshot.planPath)) {
1928
1730
  execution = null;
1929
1731
  updateStatusWidget(ctx);
1930
1732
  return;
1931
1733
  }
1734
+ // Re-derive the task tree from the plan file (fresh parse) merged with
1735
+ // the snapshot's persisted statuses — a stale parse cannot freeze progress.
1736
+ let tasks: TaskView[];
1737
+ let items: CheckItem[] = snapshot.items.map((item) => ({ ...item }));
1738
+ let planTasks = snapshot.planTasks;
1739
+ try {
1740
+ const planText = fs.readFileSync(snapshot.planPath, "utf8");
1741
+ planTasks = parsePlanTasks(planText);
1742
+ const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
1743
+ tasks = buildTaskView(planTasks, snapshotProgress);
1744
+ items = parseChecklist(planText);
1745
+ } catch {
1746
+ tasks = snapshot.tasks;
1747
+ }
1932
1748
  execution = {
1933
1749
  planPath: snapshot.planPath,
1934
- items: snapshot.items.map((item) => ({ ...item })),
1750
+ items,
1751
+ planTasks,
1752
+ tasks,
1753
+ legacyPlan: snapshot.legacyPlan,
1935
1754
  startedAt: snapshot.startedAt,
1936
1755
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
1937
- implItems: snapshot.implItems ?? [],
1938
- implStatus: { ...(snapshot.implStatus ?? {}) },
1939
- // D-008: chrome language is re-resolved at restore time from the
1940
- // CURRENT config rather than trusted from the snapshot, so a
1941
- // `plans set-language` change survives restarts.
1942
1756
  uiLanguage: resolveUiLanguage(ctx.cwd),
1943
- currentI: snapshot.currentI ?? inferCurrentI(snapshot.implItems, snapshot.items, snapshot.implStatus),
1944
- goalWait: snapshot.goalWait
1945
- ? { ...snapshot.goalWait }
1946
- : { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
1757
+ stall: { ...snapshot.stall, lastSnapshot: null, rounds: 0 },
1758
+ audit: { rounds: snapshot.audit?.rounds ?? 0, failed: snapshot.audit?.failed ?? [], running: false },
1759
+ auditLatch: { auditedThisSettle: false, activity: 0 },
1947
1760
  };
1948
- // Distrust the snapshot's implItems: re-parse + re-lint from the plan
1949
- // file so a stale empty list (older parse or format drift at snapshot
1950
- // time) cannot freeze a fake "I 0/0" panel after a restart.
1951
- try {
1952
- const planText = fs.readFileSync(snapshot.planPath, "utf8");
1953
- execution.implItems = parseImplItems(planText);
1954
- execution.implWarning = lintImplItems(planText);
1955
- } catch {
1956
- /* plan file unreadable mid-restore: keep the snapshot values */
1957
- }
1958
- for (let i = snapshotIndex + 1; i < entries.length; i++) {
1959
- const entry = entries[i];
1960
- if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1961
- execution = null;
1962
- break;
1963
- }
1964
- const message = entry.message;
1965
- if (message && message.role === "assistant") {
1966
- const text = message.content
1967
- .filter((part) => part.type === "text")
1968
- .map((part) => part.text ?? "")
1969
- .join("\n");
1970
- applyDoneMarkers(text);
1971
- applyImplMarkers(text);
1972
- applyCurrentIMarker(text);
1973
- }
1974
- }
1975
- if (execution) {
1976
- resetGoalWaitRuntime(ctx);
1977
- // Rebind the run identity after a restart so the panel's activity row
1978
- // (and any run-status mirroring) resolves to the active run instead of
1979
- // staying null until the next startExecution (CQ1/D-005 wiring gap).
1980
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1981
- executionRunId = active?.run_id ?? null;
1982
- if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1983
- // D-010: replay may have advanced progress past the persisted baseline.
1984
- // Recompute the goal-wait markers; new progress resets the guard counters.
1985
- if (execution.goalWait) {
1986
- const markerSnapshot = goalWaitSnapshot();
1987
- if (markerSnapshot !== execution.goalWait.lastMarkers) {
1988
- execution.goalWait.lastMarkers = markerSnapshot;
1989
- execution.goalWait.noProgressRounds = 0;
1990
- execution.goalWait.waitRounds = 0;
1991
- }
1992
- }
1993
- persist(pi); // refresh snapshot so the next resume has less to rescan
1994
- if (isExecutionComplete()) {
1995
- // Completed during the rescan: restore the planning model on the way out.
1996
- await completeExecution(pi, ctx);
1997
- }
1761
+ execution.stall.lastSnapshot = stallSnapshot();
1762
+ resetContinuationRuntime(ctx);
1763
+ const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1764
+ executionRunId = active?.run_id ?? null;
1765
+ if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1766
+ persist(ctx);
1767
+ if (pendingAudit()) {
1768
+ // Terminal tasks without a passing audit: rerun the audit flow.
1769
+ latchAuditThisSettle();
1770
+ await runAuditFlow(ctx);
1998
1771
  }
1999
1772
  updateStatusWidget(ctx);
2000
1773
  }