pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
package/src/exec.ts CHANGED
@@ -1,10 +1,18 @@
1
1
  /**
2
- * Plan-execution loop: the tracked execution mode for accepted plans.
2
+ * Plan-execution loop (v0.6.1): the tracked execution mode for accepted
3
+ * plans, driven by the plan's task tree.
3
4
  *
4
5
  * When the user approves the execution handoff, the extension switches into
5
- * execution mode: every agent turn is injected with the remaining verifier
6
- * checklist, assistant messages are scanned for [DONE:VC-xxx] markers, and
7
- * progress is reported through the bottom status bar until every item passes.
6
+ * execution mode: every agent turn is injected with the current wave and
7
+ * remaining tasks, task progress flows in exclusively through the
8
+ * `plans_update_task` tool (status + evidence), the task dashboard shows
9
+ * live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
10
+ * tree), a stall watchdog pauses the run when consecutive rounds produce no
11
+ * task-state change, and when every task reaches a terminal state an
12
+ * independent completion auditor verifies the plan's verification checks —
13
+ * failed checks roll their covered tasks back to pending (audit-flow-only
14
+ * channel), and three failed rounds pause for the user (bounded stopped
15
+ * termination under auto-approve/headless).
8
16
  */
9
17
 
10
18
  import * as fs from "node:fs";
@@ -36,9 +44,10 @@ import {
36
44
  type VccCompactionBuildResult,
37
45
  type VccCompactionStats,
38
46
  } from "./compaction.ts";
39
- import { getRun, lintPlanIntoNotices, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
40
- import { execChrome, resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
47
+ import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
48
+ import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
41
49
  import { bindRun, resolveActiveRun } from "./run-context.ts";
50
+ import type { SubagentProgressEvent } from "./subagent.ts";
42
51
  import { OwnershipError } from "./run-ownership.ts";
43
52
  import {
44
53
  applyExecutionApproved,
@@ -46,82 +55,80 @@ import {
46
55
  applyExecutionHeadChanged,
47
56
  applyExecutionProgress,
48
57
  applyExecutionStopped,
49
- applyQuestionAsked,
50
- applyReviewRoundStarted,
51
58
  createCheckpoint,
52
59
  loadCheckpoint,
53
60
  mutateCheckpoint,
54
- StaleCheckpointError,
55
61
  planIdentityOf,
56
62
  resolveHeadAt,
57
63
  resolveWorktreeRoot,
58
64
  sha256File,
65
+ StaleCheckpointError,
59
66
  type ExecutionApproval,
67
+ type WorkflowCheckpoint,
60
68
  } from "./workflow-state.ts";
61
69
  import { graphBlockForExecutor } from "./code-graph/prompts.ts";
62
- import {
63
- PANEL_WIDGET_KEY,
64
- derivePanelModel,
65
- deriveNextAction,
66
- deriveImplReviewLoopModel,
67
- formatImplReviewLoopSummaryLine,
68
- formatPanelSummaryLine,
69
- renderImplReviewLoopLines,
70
- renderPanelLines,
71
- themeImplReviewLoopLines,
72
- themePanelLines,
73
- } from "./panel.ts";
74
70
  import { resolveGraphMode } from "./code-graph/mode.ts";
75
71
  import {
76
- TERMINATION_QUESTION,
77
- TERMINATION_OPTIONS,
78
- TERMINATION_RECORDING_INSTRUCTIONS,
79
- defaultImplReviewers,
80
- implReviewerCountPromptLine,
81
- renderTerminationOptions,
82
- } from "./termination-prompt.ts";
83
- import {
84
- extractCoverage,
85
- latestPlanVersion,
86
72
  parseChecklist,
87
- parseImplItems,
88
- resolveImplStatuses,
89
- scanDoneMarkers,
90
- scanImplMarkers,
91
- scanCurrentIMarkers,
92
- resolveCurrentI,
93
- inferCurrentI,
94
- lintImplItems,
73
+ parsePlanTasks,
74
+ flattenTasks,
95
75
  type CheckItem,
96
- type ImplItem,
97
- type ImplMarkerState,
76
+ type PlanTasks,
98
77
  } from "./plan.ts";
78
+ import {
79
+ allTasksTerminal,
80
+ auditRollbackSet,
81
+ auditableChecks,
82
+ buildTaskView,
83
+ currentTask,
84
+ flattenTaskViews,
85
+ taskIsTerminal,
86
+ taskProgress,
87
+ taskProgressMap,
88
+ type TaskProgressMap,
89
+ type TaskView,
90
+ } from "./tasks.ts";
91
+ import { isAutoApproveEnabled as isAutoApproveEnabledLocal } from "./auto-approve.ts";
92
+ import {
93
+ DASHBOARD_WIDGET_KEY,
94
+ deriveDashboardModel,
95
+ formatDashboardSummaryLine,
96
+ formatElapsed,
97
+ renderDashboardLines,
98
+ renderDashboardTreeLines,
99
+ } from "./dashboard.ts";
100
+ import { AUDIT_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit } from "./auditor.ts";
101
+ import { messaging } from "./messaging.ts";
99
102
 
100
103
  export interface ExecState {
101
104
  planPath: string;
105
+ /** Verification checks (VC-###) — the audit's contract. */
102
106
  items: CheckItem[];
107
+ /** Parsed plan task model (kept for re-deriving the view). */
108
+ planTasks: PlanTasks;
109
+ /** Live task tree (the single source of progress). */
110
+ tasks: TaskView[];
111
+ /** True when the plan parsed through the legacy I-### fallback. */
112
+ legacyPlan: boolean;
103
113
  startedAt: string;
104
114
  usage: { inToks: number; outToks: number };
105
- implItems?: ImplItem[];
106
- implStatus?: Record<string, ImplMarkerState>;
107
- /** Plan-lint warning backing the panel's implWarning line. */
108
- implWarning?: string | null;
109
- /** Chrome language for panel/status strings (issue #3); undefined → "en". */
115
+ /** Chrome language for panel/status strings; undefined → "en". */
110
116
  uiLanguage?: UiLanguage;
111
- currentI?: string;
112
- goalWait?: GoalWaitState;
117
+ /** Stall watchdog (v0.6.1): consecutive settled rounds without a task
118
+ * status change; auto-pause at the cap. */
119
+ stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
120
+ /** Completion-audit bookkeeping. */
121
+ audit: { rounds: number; failed: string[]; running: boolean };
122
+ /** Per-settle audit latch (v0.7.1): a settled round fires the completion
123
+ * audit at most once, so the turn_end / agent_before_settle / resume entry
124
+ * points cannot double-consume a round when several land in one settle.
125
+ * Created on demand by auditLatchOf(); every construction path may omit it. */
126
+ auditLatch?: { auditedThisSettle: boolean; activity: number };
113
127
  }
114
128
 
115
129
  /**
116
- * D-008 (issue #3): re-resolve the chrome language and repaint the panel and
117
- * status bar. Called by the plans tool right after a successful
118
- * `set-language` so an executing run switches language without a restart
119
- * (and without reading config on every render tick).
120
- *
121
- * Capability guard (implementation review F-001): partial contexts (some
122
- * command/test harnesses expose only notify/select/input) may lack
123
- * setStatus/theme — the refresh must stay a no-op there instead of throwing
124
- * into the caller's error path (updateStatusWidget assumes a full ui).
130
+ * D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
131
+ * and status bar right after a `set-language` change.
125
132
  */
126
133
  export function refreshUiLanguage(ctx: ExtensionContext): void {
127
134
  if (execution) execution.uiLanguage = resolveUiLanguage(ctx.cwd);
@@ -129,43 +136,35 @@ export function refreshUiLanguage(ctx: ExtensionContext): void {
129
136
  updateStatusWidget(ctx);
130
137
  }
131
138
 
132
- export interface GoalWaitState {
133
- noProgressRounds: number;
134
- waitRounds: number;
135
- /** Marker/progress snapshot of the last goal-wait round; null = baseline not set. */
136
- lastMarkers: string | null;
137
- paused: boolean;
138
- pausedReason?: string;
139
- }
140
-
141
- const GOAL_WAIT_MAX_NO_PROGRESS = 3;
142
- const GOAL_WAIT_MAX_WAITING = 6;
139
+ /** Consecutive no-progress rounds before the watchdog pauses (D-021). */
140
+ const STALL_MAX_ROUNDS = 3;
143
141
 
144
142
  let execution: ExecState | null = null;
145
143
 
146
- export const GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
144
+ export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
145
+ /** Legacy v0.6.0 continuation message type — filtered on restore. */
146
+ const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
147
147
 
148
- interface GoalWaitRuntime {
148
+ interface ContinuationRuntime {
149
149
  owner: ExecState;
150
150
  session: ExtensionContext["sessionManager"];
151
151
  handled: boolean;
152
152
  stopReason?: string;
153
- text: string;
154
153
  wakeId?: string;
155
154
  }
156
155
 
157
- // Dispatch identity belongs to a live session, never to a persisted checklist.
158
- let goalWaitRuntime: GoalWaitRuntime | null = null;
156
+ // Dispatch identity belongs to a live session, never to a persisted state.
157
+ let continuationRuntime: ContinuationRuntime | null = null;
159
158
 
160
- function resetGoalWaitRuntime(ctx: ExtensionContext): void {
161
- goalWaitRuntime = execution
162
- ? { owner: execution, session: ctx.sessionManager, handled: false, text: "" }
159
+ function resetContinuationRuntime(ctx: ExtensionContext): void {
160
+ continuationRuntime = execution
161
+ ? { owner: execution, session: ctx.sessionManager, handled: false }
163
162
  : null;
164
163
  }
165
164
 
166
- function currentGoalWaitRuntime(ctx: ExtensionContext): GoalWaitRuntime | null {
167
- return goalWaitRuntime?.owner === execution && goalWaitRuntime.session === ctx.sessionManager
168
- ? goalWaitRuntime
165
+ function currentContinuationRuntime(ctx: ExtensionContext): ContinuationRuntime | null {
166
+ return continuationRuntime?.owner === execution && continuationRuntime.session === ctx.sessionManager
167
+ ? continuationRuntime
169
168
  : null;
170
169
  }
171
170
 
@@ -179,17 +178,14 @@ export function consumePendingExecutionFlush(): boolean {
179
178
  return pending;
180
179
  }
181
180
 
182
- function requestExecutionFlush(_pi: ExtensionAPI, _ctx: ExtensionContext): void {
183
- // Unconditional defer. turn_end fires mid-run in a gap between agent
184
- // operations where isIdle() reads true; persistence happens only at the
185
- // drain points: agent_settled, the next before_agent_start, and stop/complete.
181
+ function requestExecutionFlush(): void {
186
182
  pendingExecutionFlush = true;
187
183
  }
188
184
 
189
- export function drainExecutionFlush(pi: ExtensionAPI, ctx: ExtensionContext): void {
185
+ export function drainExecutionFlush(ctx: ExtensionContext): void {
190
186
  if (!execution || !pendingExecutionFlush) return;
191
187
  pendingExecutionFlush = false;
192
- persist(pi);
188
+ persist(ctx);
193
189
  updateStatusWidget(ctx);
194
190
  }
195
191
 
@@ -203,19 +199,22 @@ export interface CheckpointExecutionLoad {
203
199
  doneVcIds?: string[];
204
200
  reverifyAll?: boolean;
205
201
  pausedReason?: string;
202
+ /** v0.6.1: true when the checkpoint parsed through the legacy fallback. */
203
+ legacyPlan?: boolean;
204
+ /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
205
+ * detected — resume requires a fresh handoff approval. */
206
+ legacyDelegate?: boolean;
206
207
  error?: string;
207
208
  }
208
209
 
209
210
  /**
210
- * Shared restore primitive (I-005/I-006): load the executing state from a run
211
- * checkpoint into THIS session. Authorization is kept only when the recorded
212
- * approval matches the current plan digest; a HEAD change keeps the
213
- * authorization but re-verifies previously verified VCs (D-011/F-001).
214
- * F-002: the loaded state is persisted to the current session IMMEDIATELY so
215
- * session_start/session_tree restore paths cannot silently clear it.
211
+ * Shared restore primitive: load the executing state from a run checkpoint
212
+ * into THIS session. Authorization is kept only when the recorded approval
213
+ * matches the current plan digest; a HEAD change keeps the authorization but
214
+ * re-opens previously closed tasks (D-023: reverifyAll → task statuses are
215
+ * dropped and re-run).
216
216
  */
217
217
  export function loadExecutionFromCheckpoint(
218
- pi: ExtensionAPI,
219
218
  ctx: ExtensionContext,
220
219
  runId: string,
221
220
  ): CheckpointExecutionLoad {
@@ -229,9 +228,6 @@ export function loadExecutionFromCheckpoint(
229
228
  return { status: "plan-missing", error: planPath ? `plan file vanished: ${planPath}` : "checkpoint has no plan identity" };
230
229
  }
231
230
  const planText = fs.readFileSync(planPath, "utf8");
232
- // F-002 (implementation review): the recorded plan identity is over BYTES —
233
- // an in-place edit at the same path must not inherit the authorization or
234
- // the verified VCs. Refuse the load and require a fresh handoff.
235
231
  if (sha256File(planPath) !== cp.plan.sha256) {
236
232
  return {
237
233
  status: "plan-mismatch" as const,
@@ -240,61 +236,107 @@ export function loadExecutionFromCheckpoint(
240
236
  }
241
237
  const items = parseChecklist(planText);
242
238
  if (items.length === 0) {
243
- return { status: "plan-missing", error: `${planPath} has no parsable verifier checklist` };
239
+ return { status: "plan-missing", error: `${planPath} has no parsable verification checks` };
240
+ }
241
+ const planTasks = parsePlanTasks(planText);
242
+ if (planTasks.tasks.length === 0) {
243
+ return { status: "plan-missing", error: `${planPath} has no parsable tasks (## Tasks or legacy ## Implementation Items)` };
244
244
  }
245
- const implItems = parseImplItems(planText);
246
- const doneIds = new Set(cp.execution.doneVcIds);
247
- // D-011/F-001: an unchanged plan digest keeps the recorded authorization;
248
- // a changed HEAD under it forces re-verification of previously verified VCs.
249
- // F-006 (implementation review): an approval without a resolvable HEAD
250
- // recorded an unverifiable code state — re-verify instead of trusting.
251
245
  const headNow = resolveHeadAt(ctx.cwd);
252
246
  const headUnverifiable = cp.execution.approval === null || cp.execution.approval.headAtApproval === null;
253
247
  const headChanged =
254
248
  cp.execution.approval !== null &&
255
249
  cp.execution.approval.headAtApproval !== null &&
256
250
  cp.execution.approval.headAtApproval !== headNow;
251
+ // D-023: reverifyAll re-opens every closed task (statuses dropped); a
252
+ // normal restore replays the persisted task progress map.
257
253
  const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
254
+ const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
255
+ const tasks = buildTaskView(planTasks, progress);
256
+ // An audit-cap pause grants a fresh audit budget on restore (mirrors
257
+ // resumeGoalWaitIfPaused) so the first turn can actually re-audit.
258
+ const wasAuditCapPause = (cp.execution.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
258
259
  if (!reverifyAll) {
259
- for (const item of items) {
260
- if (doneIds.has(item.id)) item.done = true;
260
+ for (const id of cp.execution.doneVcIds) {
261
+ const item = items.find((candidate) => candidate.id === id);
262
+ if (item) item.done = true;
261
263
  }
262
264
  }
263
265
  execution = {
264
266
  planPath,
265
267
  items,
268
+ planTasks,
269
+ tasks,
270
+ legacyPlan: planTasks.legacy,
266
271
  startedAt: utcNow(),
267
272
  usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
268
- implItems,
269
- implStatus: { ...cp.execution.implStatus },
270
- implWarning: lintImplItems(planText),
271
273
  uiLanguage: resolveUiLanguage(ctx.cwd),
272
- currentI: cp.execution.currentI,
273
- goalWait: {
274
- noProgressRounds: 0,
275
- waitRounds: 0,
276
- lastMarkers: null,
277
- paused: cp.execution.pausedReason !== undefined,
278
- pausedReason: cp.execution.pausedReason,
274
+ // D-020: a paused legacy (or stopped) execution rebuilds unpaused —
275
+ // the resume itself is the user's intent; the reason is surfaced in
276
+ // the resume brief instead. An audit-cap pause additionally grants a
277
+ // fresh audit budget here (mirrors resumeGoalWaitIfPaused), so the
278
+ // first turn after a cross-session resume can actually re-audit.
279
+ stall: {
280
+ rounds: 0,
281
+ lastSnapshot: null,
282
+ paused: false,
283
+ pausedReason: undefined,
284
+ },
285
+ audit: {
286
+ rounds: wasAuditCapPause ? 0 : (cp.execution.audit?.rounds ?? 0),
287
+ failed: [],
288
+ running: false,
279
289
  },
290
+ auditLatch: { auditedThisSettle: false, activity: 0 },
280
291
  };
292
+ execution.stall.lastSnapshot = stallSnapshot();
281
293
  executionRunId = runId;
282
294
  bindRun(ctx.sessionManager, ctx.cwd, runId);
283
- resetGoalWaitRuntime(ctx);
284
- pendingExecutionFlush = false; // restored state: no inherited flush debt
295
+ resetContinuationRuntime(ctx);
296
+ pendingExecutionFlush = false;
285
297
  resetExecutionCompactionState(ctx);
298
+ if (wasAuditCapPause) {
299
+ withExecutionCheckpoint(ctx, (cp2) =>
300
+ applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
301
+ );
302
+ }
303
+ // D-020: an orphaned v0.6.0 delegated executor never survives a restart.
304
+ // Its checkpoint delegate marker REFUSES the direct load — the run must
305
+ // re-enter through the execution handoff so the C-006 approval gate
306
+ // applies; execution restarts from the first task after re-approval.
307
+ const legacyDelegate = cp.execution.delegate !== undefined;
308
+ if (legacyDelegate) {
309
+ try {
310
+ ctx.ui.notify?.(
311
+ "pi-plans: this run was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Re-approve via /plans-execute; execution restarts from the first task (the 0.6.0 progress record cannot map onto the task tree).",
312
+ "warning",
313
+ );
314
+ } catch {
315
+ /* best-effort */
316
+ }
317
+ // Refuse the load: no execution state may activate without the fresh
318
+ // C-006 handoff approval.
319
+ execution = null;
320
+ executionRunId = null;
321
+ return {
322
+ status: "no-execution",
323
+ legacyDelegate: true,
324
+ error: "orphaned v0.6.0 delegated executor; re-approve via /plans-execute (execution restarts from the first task)",
325
+ };
326
+ }
286
327
  if (headChanged) {
287
328
  withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
288
329
  }
289
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
290
- persist(pi); // F-002: immediate session snapshot
330
+ persist(ctx);
291
331
  updateStatusWidget(ctx);
292
332
  return {
293
333
  status: "loaded",
294
334
  planPath,
295
- doneVcIds: [...doneIds],
335
+ doneVcIds: [...cp.execution.doneVcIds],
296
336
  reverifyAll,
297
337
  pausedReason: cp.execution.pausedReason,
338
+ legacyPlan: planTasks.legacy,
339
+ legacyDelegate,
298
340
  };
299
341
  }
300
342
 
@@ -308,7 +350,6 @@ interface ExecutionCompactionState {
308
350
  lastSuccessfulUsagePercent: number | null;
309
351
  lastSuccessfulAt: string | null;
310
352
  rearmPending: boolean;
311
- /** Terminal failure metadata is retained for diagnostics, not proactive retry. */
312
353
  terminalBackoffTokens: number | null;
313
354
  pendingStats: VccCompactionStats | null;
314
355
  pendingFollowUpPrompt: string | null;
@@ -368,63 +409,21 @@ export function handleExecutionTurnCompaction(ctx: ExtensionContext): void {
368
409
  consumeExecutionCompactionResumeGuard(ctx);
369
410
  }
370
411
 
371
- export function computeExecutionProgress(execution: ExecState): { done: number; total: number } {
372
- const implItems = execution.implItems ?? [];
373
- if (implItems.length) {
374
- const statuses = resolveImplStatuses(implItems, execution.items, execution.implStatus);
375
- const counted = implItems.filter((impl) =>
376
- execution.items.some((item) => extractCoverage(item.text).includes(impl.id)),
377
- );
378
- const total = counted.length > 0 ? counted.length : implItems.length;
379
- const vcDone = counted.filter((impl) => statuses[impl.id] === "vc-passed").length;
380
- const currentIndex = execution.currentI
381
- ? implItems.findIndex((impl) => impl.id === execution.currentI)
382
- : -1;
383
- return {
384
- done: Math.min(total, Math.max(vcDone, currentIndex < 0 ? 0 : currentIndex)),
385
- total,
386
- };
387
- }
388
- return {
389
- done: execution.items.filter((item) => item.done).length,
390
- total: execution.items.length,
391
- };
392
- }
393
-
394
- function formatElapsed(startedAt: string): string {
395
- const total = Math.max(0, Math.floor((Date.now() - Date.parse(startedAt)) / 1000));
396
- const h = String(Math.floor(total / 3600)).padStart(2, "0");
397
- const m = String(Math.floor((total % 3600) / 60)).padStart(2, "0");
398
- const sec = String(total % 60).padStart(2, "0");
399
- return `${h}:${m}:${sec}`;
400
- }
401
-
402
412
  function formatToks(tokens: number): string {
403
413
  const n = Math.max(0, Math.round(tokens));
404
414
  return n < 1000 ? String(n) : `${(n / 1000).toFixed(1)}k`;
405
415
  }
406
416
 
407
417
  export function formatExecutionStatusLine(execution: ExecState): string {
408
- const progress = computeExecutionProgress(execution);
409
- let line = `⌛ plans ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
410
- const goalWait = execution.goalWait;
411
- if (goalWait?.paused) {
412
- line += ` · ⏸ goal-wait paused (${goalWait.pausedReason ?? "paused"})`;
413
- } else if (goalWait && (goalWait.noProgressRounds > 0 || goalWait.waitRounds > 0)) {
414
- line += execChrome(execution.uiLanguage ?? "en").goalWait(goalWait.noProgressRounds, goalWait.waitRounds);
418
+ const progress = taskProgress(execution.tasks);
419
+ let line = `⌛ plans tasks ${progress.done}/${progress.total}: spent ${formatElapsed(execution.startedAt)} · ${formatToks(execution.usage.inToks)} in-toks · ${formatToks(execution.usage.outToks)} out-toks`;
420
+ if (execution.stall.paused) {
421
+ line += ` · ⏸ paused (${execution.stall.pausedReason ?? "stalled"})`;
415
422
  }
416
423
  return line;
417
424
  }
418
425
 
419
- /** Approximation for "a subprocess is pending" (matches the exec loop's
420
- * `/waiting for/` backoff heuristic; F-008). Passed explicitly into the
421
- * shared model so the panel, status line and injection text agree. */
422
- export function executionIsWaiting(execution: ExecState): boolean {
423
- const gw = execution.goalWait;
424
- return gw !== undefined && !gw.paused && gw.waitRounds > 0;
425
- }
426
-
427
- /** Resolve the run topic for panel headers (falls back to the run id). */
426
+ /** Resolve the run topic for dashboard headers (falls back to the run id). */
428
427
  function panelTopic(ctx: ExtensionContext): string {
429
428
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
430
429
  if (active) {
@@ -435,146 +434,77 @@ function panelTopic(ctx: ExtensionContext): string {
435
434
  return "pi-plans";
436
435
  }
437
436
 
438
- /** Run info for the panel activity line (CQ1/D-005). Read by the in-flight
439
- * executionRunId so a stale/missing run record degrades to null (the model
440
- * then falls back to the bare phase word) instead of showing another run's
441
- * status. */
442
- function panelRunInfo(ctx: ExtensionContext): { status: string; created_at: string; updated_at: string } | null {
443
- if (!executionRunId) return null;
444
- // D-005 pointer-consistency: if the workdir's active pointer has moved to
445
- // another run (second session / external CLI mutation) while this
446
- // execution is live, the activity row degrades to the phase word rather
447
- // than mixing the new active run's topic (header) with the old run's
448
- // status (impl-review r1 F-001).
449
- if (resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id !== executionRunId) return null;
450
- const run = getRun(ctx.cwd, executionRunId);
451
- if (!run) return null;
452
- return { status: run.status, created_at: run.created_at, updated_at: run.updated_at };
453
- }
437
+ let dashboardRegistered = false;
438
+ let dashboardExpanded = false;
454
439
 
455
- let panelRegistered = false;
456
- let loopPanelRegistered = false;
440
+ /** Toggle the dashboard's expanded tree view (Ctrl+Shift+T). */
441
+ export function toggleDashboardExpanded(ctx: ExtensionContext): void {
442
+ dashboardExpanded = !dashboardExpanded;
443
+ dashboardRegistered = false; // force re-registration with the new mode
444
+ updateStatusWidget(ctx);
445
+ }
457
446
 
458
- /** Live implementation-review loop state for the panel: the widget stays
459
- * alive while the active run is done BUT its checkpoint is still in the
460
- * implementation-review phase (D-3). Read fresh on every call so post-write
461
- * redraws (index.ts turn-end updateStatusWidget) never show stale rounds
462
- * (D-8). Returns null once the phase flips to completed. */
463
- function implReviewLoopState(
464
- ctx: ExtensionContext,
465
- ): { topic: string; review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } } | null {
466
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
467
- if (!active) return null;
468
- if (getRun(ctx.cwd, active.run_id)?.status !== "done") return null;
469
- const load = loadCheckpoint(ctx.cwd, active.run_id);
470
- if (load.status !== "ok" || load.checkpoint.phase !== "implementation-review") return null;
471
- return { topic: panelTopic(ctx), review: load.checkpoint.implementationReview };
447
+ export function isDashboardExpanded(): boolean {
448
+ return dashboardExpanded;
472
449
  }
473
450
 
474
451
  /**
475
- * Register/update the fixed tasks' status panel (aboveEditor widget) for the
476
- * current execution, or unregister it when execution is gone. The panel and
477
- * the bottom status line share the same pure model (D-003/D-014/D-015). The
478
- * widget uses the factory form and reads the LIVE theme via `ui.theme` inside
479
- * render(width) — per-line width math happens on plain text first, then the
480
- * current theme is applied, so theme hot-swaps and resize never produce stale
481
- * colors or wrapped rows (F-004).
452
+ * Register/update the task dashboard (aboveEditor widget) for the current
453
+ * execution, or unregister it when execution is gone. The compact and the
454
+ * expanded tree view share one widget key so they never stack.
482
455
  */
483
456
  function updatePanelWidget(ctx: ExtensionContext): void {
484
- // Capability guard: older Pi hosts and test harness mocks may not expose
485
- // setWidget (the panel is a UI nicety, never a correctness dependency).
486
457
  if (!ctx.hasUI || typeof ctx.ui.setWidget !== "function") return;
487
- // Implementation-review loop widget (D-3): keeps the panel alive after
488
- // execution ends while the loop is live; unregisters when the phase
489
- // completes. Lives under the same widget key so the two widgets never
490
- // stack.
491
- const loop = execution === null ? implReviewLoopState(ctx) : null;
492
- if (execution === null && loop) {
493
- if (!loopPanelRegistered) {
494
- ctx.ui.setWidget(
495
- PANEL_WIDGET_KEY,
496
- (ui, _theme) => ({
497
- render(width: number) {
498
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
499
- // D-8 freshness: re-read the loop state per render; a phase flip to
500
- // completed renders an empty box until the next turn-end refresh
501
- // unregisters it (index.ts always calls updateStatusWidget then).
502
- const live = implReviewLoopState(ctx);
503
- if (!live) return [];
504
- const model = deriveImplReviewLoopModel(live.topic, live.review);
505
- const lines = renderImplReviewLoopLines(model, width);
506
- return theme ? themeImplReviewLoopLines(lines, theme as never) : lines;
507
- },
508
- }),
509
- { placement: "aboveEditor" },
510
- );
511
- loopPanelRegistered = true;
512
- }
513
- } else if (loopPanelRegistered) {
514
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
515
- loopPanelRegistered = false;
516
- }
517
458
  if (!execution) {
518
- if (panelRegistered) {
519
- ctx.ui.setWidget(PANEL_WIDGET_KEY, undefined);
520
- panelRegistered = false;
459
+ if (dashboardRegistered) {
460
+ ctx.ui.setWidget(DASHBOARD_WIDGET_KEY, undefined);
461
+ dashboardRegistered = false;
521
462
  }
522
463
  return;
523
464
  }
524
- if (!panelRegistered) {
525
- ctx.ui.setWidget(
526
- PANEL_WIDGET_KEY,
527
- (ui, _theme) => ({
528
- render(width: number) {
529
- const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
530
- const current = execution;
531
- if (!current) return [];
532
- // F-004 (impl review r1): recompute the topic per render so a
533
- // cross-run restart without an intervening unregister cannot
534
- // show a stale box header.
535
- const model = derivePanelModel(current, panelTopic(ctx), executionIsWaiting(current), panelRunInfo(ctx));
536
- const lines = renderPanelLines(model, width);
537
- // Uniform-gray frame: │ borders never inherit the line color;
538
- // accents live between the borders only (themePanelLines).
539
- return theme ? themePanelLines(lines, model, theme) : lines;
540
- },
541
- }),
542
- { placement: "aboveEditor" },
543
- );
544
- panelRegistered = true;
545
- } else {
546
- // Factory components are re-created on every registration; content
547
- // updates flow through the closure reads at render time, so a no-op
548
- // re-set is unnecessary. Trigger one re-render via a cheap status
549
- // touch is NOT used — event-driven flush only (D-013).
550
- }
465
+ if (dashboardRegistered) return;
466
+ ctx.ui.setWidget(
467
+ DASHBOARD_WIDGET_KEY,
468
+ (ui, _theme) => ({
469
+ render(width: number) {
470
+ const theme = (ui as { theme?: { fg(color: string, text: string): string } }).theme ?? _theme;
471
+ const current = execution;
472
+ if (!current) return [];
473
+ const model = deriveDashboardModel(panelTopic(ctx), current.tasks, current.items, {
474
+ paused: current.stall.paused,
475
+ pausedReason: current.stall.pausedReason,
476
+ auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
477
+ auditFailed: current.audit.failed,
478
+ startedAt: current.startedAt,
479
+ usage: current.usage,
480
+ });
481
+ const lines = dashboardExpanded
482
+ ? renderDashboardTreeLines(model, width, theme)
483
+ : renderDashboardLines(model, width, theme);
484
+ return lines;
485
+ },
486
+ }),
487
+ { placement: "aboveEditor" },
488
+ );
489
+ dashboardRegistered = true;
551
490
  }
552
491
 
553
492
  export function updateStatusWidget(ctx: ExtensionContext): void {
554
493
  updatePanelWidget(ctx);
555
- if (execution) {
556
- // D-015: the status line derives from the same panel model.
557
- const line = formatPanelSummaryLine(
558
- derivePanelModel(execution, panelTopic(ctx), executionIsWaiting(execution), panelRunInfo(ctx)),
559
- );
560
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
494
+ if (execution && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
495
+ const model = deriveDashboardModel(panelTopic(ctx), execution.tasks, execution.items, {
496
+ paused: execution.stall.paused,
497
+ pausedReason: execution.stall.pausedReason,
498
+ auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
499
+ auditFailed: execution.audit.failed,
500
+ });
501
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
561
502
  return;
562
503
  }
563
504
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
564
- if (active) {
565
- // Idle indicator depends on the run's lifecycle, not just its existence:
566
- // done reads as finished, abandoned as closed, stopped/accepted as paused.
567
- const status = getRun(ctx.cwd, active.run_id)?.status;
505
+ if (active && typeof ctx.ui.setStatus === "function" && ctx.ui.theme) {
506
+ const status = getRun(ctx.cwd, active.run_id)?.status ?? latestRun(ctx.cwd)?.status;
568
507
  if (status === "done") {
569
- // D-3/D-015: while the implementation-review loop is live (checkpoint
570
- // still in the implementation-review phase), the status line mirrors
571
- // the loop box model instead of a bare "(done)".
572
- const loop = implReviewLoopState(ctx);
573
- if (loop) {
574
- const line = formatImplReviewLoopSummaryLine(deriveImplReviewLoopModel(loop.topic, loop.review));
575
- ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", line));
576
- return;
577
- }
578
508
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${active.run_id} (done)`));
579
509
  return;
580
510
  }
@@ -590,44 +520,63 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
590
520
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("warning", `⌛ plans: ${active.run_id}`));
591
521
  return;
592
522
  }
523
+ if (status === "executing") {
524
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
525
+ return;
526
+ }
593
527
  if (status === "planning") {
594
- // Planning phase: 💬 while still in Q&A, 📝 once a PLAN draft exists
595
- // — kept until execution starts (then ⌛ takes over).
596
- const emoji = latestPlanVersion(active.artifact_dir) ? "📝" : "💬";
528
+ const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
597
529
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
598
530
  return;
599
531
  }
600
- // unknown status: no indicator.
601
532
  }
602
- ctx.ui.setStatus("pi-plans", undefined);
533
+ if (typeof ctx.ui?.setStatus === "function" && ctx.ui.theme) {
534
+ const terminalLatest = latestRun(ctx.cwd);
535
+ if (terminalLatest && TERMINAL_RUN_STATUSES.has(terminalLatest.status)) {
536
+ if (terminalLatest.status === "done") {
537
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("success", `🎯 plans: ${terminalLatest.run_id} (done)`));
538
+ } else {
539
+ ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("error", `🚫 plans: ${terminalLatest.run_id}`));
540
+ }
541
+ return;
542
+ }
543
+ ctx.ui.setStatus("pi-plans", undefined);
544
+ }
545
+ }
546
+
547
+ function parseLatestPlanExists(artifactDir: string): boolean {
548
+ try {
549
+ const names = fs.readdirSync(artifactDir);
550
+ return names.some((name) => /^PLAN_v\d+\.(md|markdown)$/i.test(name));
551
+ } catch {
552
+ return false;
553
+ }
603
554
  }
604
555
 
605
- function persist(pi: ExtensionAPI): void {
556
+ function persist(ctx: ExtensionContext): void {
606
557
  if (!execution) return;
607
- pi.appendEntry("pi-plans-exec", {
558
+ messaging().appendEntry("pi-plans-exec", {
608
559
  planPath: execution.planPath,
609
560
  items: execution.items,
561
+ planTasks: execution.planTasks,
562
+ tasks: execution.tasks,
563
+ legacyPlan: execution.legacyPlan,
610
564
  startedAt: execution.startedAt,
611
565
  usage: execution.usage,
612
- implItems: execution.implItems,
613
- implStatus: execution.implStatus,
614
- implWarning: execution.implWarning ?? null,
615
- currentI: execution.currentI,
616
- goalWait: execution.goalWait,
566
+ stall: execution.stall,
567
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
617
568
  });
618
569
  }
619
570
 
620
571
  /** Checkpoint bookkeeping for the executing run; best-effort for legacy runs
621
- * without checkpoints (their cross-session resume degrades to R-008 rules). */
622
- function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./workflow-state.ts").WorkflowCheckpoint) => import("./workflow-state.ts").WorkflowCheckpoint): void {
572
+ * without checkpoints. */
573
+ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: WorkflowCheckpoint) => WorkflowCheckpoint): void {
623
574
  if (!execution) return;
624
575
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
625
576
  if (!active || active.run_id !== executionRunId) return;
626
577
  try {
627
578
  mutateCheckpoint(ctx.cwd, active.run_id, mutator);
628
579
  } catch (error) {
629
- // F-005 (implementation review): ownership loss and revision staleness
630
- // must stop the advance, not vanish into the catch block.
631
580
  if (error instanceof OwnershipError || error instanceof StaleCheckpointError) throw error;
632
581
  /* legacy run or corrupt checkpoint: session snapshot still carries the loop */
633
582
  }
@@ -635,88 +584,75 @@ function withExecutionCheckpoint(ctx: ExtensionContext, mutator: (cp: import("./
635
584
 
636
585
  let executionRunId: string | null = null;
637
586
 
587
+ export interface StartExecutionInput {
588
+ planPath: string;
589
+ planTasks: PlanTasks;
590
+ items: CheckItem[];
591
+ }
592
+
638
593
  export async function startExecution(
639
- pi: ExtensionAPI,
640
594
  ctx: ExtensionContext,
641
- planPath: string,
642
- items: CheckItem[],
643
- implItems?: ImplItem[],
595
+ input: StartExecutionInput,
644
596
  ): Promise<void> {
597
+ const tasks = buildTaskView(input.planTasks);
645
598
  execution = {
646
- planPath,
647
- items,
599
+ planPath: input.planPath,
600
+ items: input.items,
601
+ planTasks: input.planTasks,
602
+ tasks,
603
+ legacyPlan: input.planTasks.legacy,
648
604
  startedAt: utcNow(),
649
605
  usage: { inToks: 0, outToks: 0 },
650
- implItems: implItems ?? [],
651
- implStatus: {},
652
606
  uiLanguage: resolveUiLanguage(ctx.cwd),
653
- // F-001 (impl review r1): derive the plan-lint warning on the live
654
- // handoff path too, so the panel shows the ⚠ line immediately for a
655
- // zero-parse section instead of only after a checkpoint restore.
656
- implWarning: (() => {
657
- try {
658
- return lintImplItems(fs.readFileSync(planPath, "utf8"));
659
- } catch {
660
- return null;
661
- }
662
- })(),
663
- goalWait: { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
607
+ stall: { rounds: 0, lastSnapshot: null, paused: false },
608
+ audit: { rounds: 0, failed: [], running: false },
609
+ auditLatch: { auditedThisSettle: false, activity: 0 },
664
610
  };
665
- // Seed the marker baseline so the first quiet round is counted against a
666
- // real snapshot instead of counting unconditionally (F-006).
667
- if (execution.goalWait) execution.goalWait.lastMarkers = goalWaitSnapshot();
668
- resetGoalWaitRuntime(ctx);
669
- pendingExecutionFlush = false; // fresh run: no inherited flush debt
611
+ // Seed the watchdog baseline only after `execution` points at the new state
612
+ // (stallSnapshot reads the live execution).
613
+ execution.stall.lastSnapshot = stallSnapshot();
614
+ resetContinuationRuntime(ctx);
615
+ pendingExecutionFlush = false;
670
616
  resetExecutionCompactionState(ctx);
671
- persist(pi);
617
+ persist(ctx);
672
618
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
673
619
  executionRunId = active?.run_id ?? null;
674
620
  if (active) {
675
621
  bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
676
- // I-005: durable approval evidence — run + plan digest + HEAD at
677
- // approval (D-003/D-011). Sets phase executing via the state machine.
678
622
  try {
679
623
  const load = loadCheckpoint(ctx.cwd, active.run_id);
680
624
  if (load.status === "missing") {
681
625
  createCheckpoint(ctx.cwd, { runId: active.run_id, originWorkdir: ctx.cwd, workdir: ctx.cwd });
682
626
  }
683
627
  const approval: ExecutionApproval = {
684
- plan: planIdentityOf(path.resolve(planPath), 1),
628
+ plan: planIdentityOf(path.resolve(input.planPath), 1),
685
629
  worktree: resolveWorktreeRoot(ctx.cwd) ?? path.resolve(ctx.cwd),
686
630
  headAtApproval: resolveHeadAt(ctx.cwd),
687
631
  approvedAt: utcNow(),
688
632
  };
689
633
  mutateCheckpoint(ctx.cwd, active.run_id, (cp) => {
690
- // Plan refinement may not have recorded the plan identity yet.
691
634
  const withPlan = cp.plan === null ? { ...cp, plan: approval.plan } : cp;
692
- // The checkpoint may not carry accept-execute (legacy flow);
693
- // approval here came from the explicit handoff confirmation.
694
635
  const aligned = withPlan.nextAction === "accept-execute"
695
636
  ? withPlan
696
637
  : { ...withPlan, nextAction: "accept-execute" as const };
697
638
  return applyExecutionApproved(aligned, approval);
698
639
  });
699
- // Plan-lint entry point (execute handoff): a plan whose Implementation
700
- // Items section parses to zero items gets a durable run notice so the
701
- // execution panel's warning is backed by persisted evidence.
702
- lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(planPath));
640
+ lintPlanIntoNotices(ctx.cwd, active.run_id, path.resolve(input.planPath));
703
641
  } catch (error) {
704
- // F-002 (implementation review): a plan-digest mismatch between the
705
- // recorded checkpoint plan and the approval must fail closed and
706
- // visibly — never silently execute without durable approval.
707
642
  if (error instanceof StateError && /does not match/.test(error.message)) throw error;
708
643
  /* legacy/corrupt checkpoint: run status still transitions below */
709
644
  }
710
645
  try {
711
646
  setRunStatus(ctx.cwd, active.run_id, "executing");
712
647
  } catch {
713
- /* status bookkeeping is best-effort */
648
+ /* best-effort */
714
649
  }
715
650
  }
716
- pi.sendMessage(
651
+ const progress = taskProgress(tasks);
652
+ messaging().sendMessage(
717
653
  {
718
654
  customType: "pi-plans-exec-start",
719
- content: `**pi-plans: executing** \`${planPath}\` — ${items.length} verifier item(s). Progress appears in the bottom status bar; mark verified items with \`[DONE:VC-xxx]\`.`,
655
+ content: `**pi-plans: executing** \`${input.planPath}\` — ${progress.total} task(s) in ${input.planTasks.legacy ? "legacy" : "task-tree"} mode, ${input.items.length} verification check(s). Report progress with the \`plans_update_task\` tool; the dashboard tracks every task (Ctrl+Shift+T expands the tree).`,
720
656
  display: true,
721
657
  },
722
658
  { triggerTurn: false },
@@ -724,11 +660,28 @@ export async function startExecution(
724
660
  updateStatusWidget(ctx);
725
661
  }
726
662
 
727
- /** Record one assistant turn: accumulate usage and mark any completed items. */
663
+ /** Persist the live task progress (called by the task status tool). */
664
+ export function persistTaskProgress(ctx: ExtensionContext): void {
665
+ if (!execution) return;
666
+ // Any task-state change resets the stall watchdog baseline.
667
+ execution.stall.rounds = 0;
668
+ execution.stall.lastSnapshot = stallSnapshot();
669
+ withExecutionCheckpoint(ctx, (cp) =>
670
+ applyExecutionProgress(cp, {
671
+ tasks: taskProgressMap(execution!.tasks),
672
+ doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
673
+ audit: {
674
+ rounds: execution!.audit.rounds,
675
+ lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
676
+ },
677
+ }),
678
+ );
679
+ persist(ctx);
680
+ }
681
+
682
+ /** Record one assistant turn: accumulate usage only (markers are gone). */
728
683
  export function recordExecutionTurn(
729
- pi: ExtensionAPI,
730
- _ctx: ExtensionContext,
731
- _completedIds: string[],
684
+ ctx: ExtensionContext,
732
685
  usage?: { input: number; output: number },
733
686
  ): void {
734
687
  if (!execution) return;
@@ -736,102 +689,238 @@ export function recordExecutionTurn(
736
689
  execution.usage.inToks += usage.input;
737
690
  execution.usage.outToks += usage.output;
738
691
  }
739
- // I-005: mirror progress into the run checkpoint so a different session
740
- // can resume with the verified VC/I set (R-004).
741
- withExecutionCheckpoint(_ctx, (cp) =>
692
+ withExecutionCheckpoint(ctx, (cp) =>
742
693
  applyExecutionProgress(cp, {
743
- doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
744
- implStatus: implStatusSnapshot(),
745
- currentI: execution!.currentI,
746
694
  usage: usage ? { inToks: usage.input, outToks: usage.output } : undefined,
747
695
  }),
748
696
  );
749
- requestExecutionFlush(pi, _ctx);
750
- updateStatusWidget(_ctx);
697
+ requestExecutionFlush();
698
+ updateStatusWidget(ctx);
751
699
  }
752
700
 
753
- function implStatusSnapshot(): Record<string, string> {
754
- const snapshot: Record<string, string> = {};
755
- if (!execution?.implItems) return snapshot;
756
- for (const item of execution.implItems) {
757
- const state = execution.implStatus?.[item.id];
758
- if (state) snapshot[item.id] = state;
759
- }
760
- return snapshot;
701
+ /** Test hook: replace the audit subagent with a deterministic function. */
702
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
703
+
704
+ export function __setAuditRunnerForTests(
705
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
706
+ ): void {
707
+ auditRunnerForTests = runner;
761
708
  }
762
709
 
763
710
  export function registerExecutionTurnHandlers(
764
- pi: ExtensionAPI,
711
+ ext: ExtensionAPI,
765
712
  onTurnEnd?: (ctx: ExtensionContext) => Promise<void> | void,
766
713
  ): void {
767
- // The turn_end projection does not carry usage; message_end delivers the
768
- // full assistant message, so cache it here and consume it per turn.
769
714
  let lastAssistantUsage: { input: number; output: number } | null = null;
770
- pi.on("agent_start", async (_event, ctx) => {
771
- const runtime = currentGoalWaitRuntime(ctx);
715
+ ext.on("agent_start", async (_event, ctx) => {
716
+ const runtime = currentContinuationRuntime(ctx);
772
717
  if (!runtime) return;
773
718
  runtime.handled = false;
774
719
  runtime.stopReason = undefined;
775
- runtime.text = "";
720
+ // v0.7.1: a new agent run opens a new settle window (per-settle latch).
721
+ if (execution) resetSettleLatch();
776
722
  });
777
- pi.on("before_agent_start", async (_event, ctx) => {
778
- const runtime = currentGoalWaitRuntime(ctx);
723
+ ext.on("before_agent_start", async (_event, ctx) => {
724
+ const runtime = currentContinuationRuntime(ctx);
779
725
  if (runtime) runtime.wakeId = undefined;
780
726
  });
781
- pi.on("input", async (event, ctx) => {
782
- if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(pi, ctx);
727
+ ext.on("input", async (event, ctx) => {
728
+ if (event.source === "interactive" || event.source === "rpc") resumeGoalWaitIfPaused(ctx);
783
729
  });
784
- pi.on("agent_settled", async (_event, ctx) => {
785
- drainExecutionFlush(pi, ctx);
786
- maybeGoalWaitFollowUp(pi, ctx);
730
+ ext.on("agent_settled", async (_event, ctx) => {
731
+ drainExecutionFlush(ctx);
732
+ maybeContinuationFollowUp(ctx);
787
733
  });
788
- pi.on("session_shutdown", async (_event, ctx) => {
789
- drainExecutionFlush(pi, ctx);
734
+ // v0.7.1: `agent_settled` is notification-only per pi's contract, so the
735
+ // audit fallback lives on `agent_before_settle` — the final ACTIONABLE
736
+ // boundary. It is what makes a terminal-but-unaudited run self-heal with
737
+ // zero user input, instead of stranding until a manual /plans-execute.
738
+ ext.on("agent_before_settle", async (_event, ctx) => {
739
+ if (!pendingAudit()) return;
740
+ const runtime = currentContinuationRuntime(ctx);
741
+ if (runtime?.handled) return;
742
+ // The continuation wake owns this settle: if the loop already woke the
743
+ // agent to fix rolled-back work, the audit waits for the next settle
744
+ // (Q-2 single-wake guarantee) rather than emitting a second wake.
745
+ if (runtime && !allTasksTerminal(runtime.owner.tasks)) {
746
+ maybeContinuationFollowUp(ctx);
747
+ return;
748
+ }
749
+ // Fully settled and still owed an audit: run it now. The audit's own
750
+ // pass/fail message drives the rest (pass completes the run; fail
751
+ // triggers a fix turn), so no extra continuation is requested here —
752
+ // pendingAudit() is false once paused, which bounds any loop.
753
+ latchAuditThisSettle();
754
+ await runAuditFlow(ctx);
755
+ });
756
+ ext.on("session_shutdown", async (_event, ctx) => {
757
+ drainExecutionFlush(ctx);
790
758
  execution = null;
791
759
  executionRunId = null;
792
- goalWaitRuntime = null;
760
+ continuationRuntime = null;
793
761
  lastAssistantUsage = null;
794
762
  });
795
- pi.on("message_end", async (event) => {
763
+ ext.on("message_end", async (event) => {
796
764
  const message = event.message as { role?: string; usage?: { input?: number; output?: number } };
797
765
  if (message?.role === "assistant" && message.usage) {
798
766
  lastAssistantUsage = { input: message.usage.input ?? 0, output: message.usage.output ?? 0 };
799
767
  }
800
768
  });
769
+ // v0.7.1 (root cause B): the stall watchdog counted only task-status changes,
770
+ // so a round where the agent legitimately investigated (read code, gathered
771
+ // evidence) without closing a task looked identical to a dead agent. A
772
+ // SUCCESSFUL tool result is real progress; a failed/blocked tool is not, so
773
+ // an agent looping on the same error still trips the cap (F-005, Q-3).
774
+ ext.on("tool_result", async (event, _ctx) => {
775
+ if (!execution) return;
776
+ const result = event as { isError?: boolean; error?: unknown };
777
+ if (result.isError === true || result.error !== undefined) return;
778
+ auditLatchOf(execution).activity += 1; // Real progress: rebase the watchdog so this round counts as a change.
779
+ execution.stall.rounds = 0;
780
+ execution.stall.lastSnapshot = stallSnapshot();
781
+ });
801
782
 
802
- pi.on("turn_end", async (event, ctx) => {
803
- const message = event.message as { role?: string; stopReason?: string; content?: Array<{ type: string; text?: string }> };
783
+ ext.on("turn_end", async (event, ctx) => {
784
+ const message = event.message as { role?: string; stopReason?: string };
804
785
  if (!message || message.role !== "assistant") {
805
786
  updateStatusWidget(ctx);
806
787
  return;
807
788
  }
808
- const text = (message.content ?? [])
809
- .filter((part) => part.type === "text")
810
- .map((part) => part.text ?? "")
811
- .join("\n");
812
- const runtime = currentGoalWaitRuntime(ctx);
813
- if (runtime) {
814
- runtime.stopReason = message.stopReason;
815
- runtime.text = text;
816
- }
817
- const changedIds = applyDoneMarkers(text);
818
- const changedImpls = applyImplMarkers(text);
819
- const changedCurrentI = applyCurrentIMarker(text);
789
+ const runtime = currentContinuationRuntime(ctx);
790
+ if (runtime) runtime.stopReason = message.stopReason;
820
791
  const projection = (event.message as { usage?: { input?: number; output?: number } }).usage;
821
792
  const raw = projection ?? lastAssistantUsage;
822
- lastAssistantUsage = null; // consumed: never re-attribute a stale turn
793
+ lastAssistantUsage = null;
823
794
  const usage = raw ? { input: raw.input ?? 0, output: raw.output ?? 0 } : undefined;
824
- if (usage || changedIds.length > 0 || changedImpls.length > 0 || changedCurrentI) {
825
- // Attribute this turn's usage now; `[DONE]` markers still only mark completion.
826
- recordExecutionTurn(pi, ctx, changedIds, usage);
827
- }
828
- if (getExecution() && isExecutionComplete()) {
829
- await completeExecution(pi, ctx);
795
+ if (usage) recordExecutionTurn(ctx, usage);
796
+ if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
797
+ latchAuditThisSettle();
798
+ await runAuditFlow(ctx);
830
799
  }
831
800
  await onTurnEnd?.(ctx);
832
801
  });
833
802
  }
834
803
 
804
+ /** Audit flow: run the completion auditor, apply pass/rollback, and either
805
+ * complete the run, keep iterating (rollback), pause at the round cap
806
+ * (interactive), or stop at the cap (auto-approve/headless — D-022).
807
+ *
808
+ * Fail-closed completion: a run completes only when EVERY pending check was
809
+ * affirmatively passed (or resolved skipped-pass); checks the auditor failed
810
+ * to report count as failed, never as silently passed. */
811
+ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
812
+ if (!execution) return;
813
+ const ex = execution;
814
+ // Skipped-pass checks resolve without a subagent round.
815
+ for (const id of presolvedCheckIds(ex.items, ex.tasks)) {
816
+ const item = ex.items.find((candidate) => candidate.id === id);
817
+ if (item) item.done = true;
818
+ }
819
+ // Only auditable checks (with task coverage) gate completion; checks that
820
+ // cover no task can never be verified and never block or complete.
821
+ const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
822
+ if (pendingChecks.length === 0) {
823
+ await completeExecution(ctx);
824
+ return;
825
+ }
826
+ if (ex.audit.rounds >= AUDIT_MAX_ROUNDS) {
827
+ // D-022: interactive sessions pause for the user (state kept, tasks
828
+ // intact, resumable); auto-approve/headless terminates bounded.
829
+ if (isInteractiveSession(ctx)) {
830
+ pauseForStall(
831
+ ctx,
832
+ `${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
833
+ );
834
+ return;
835
+ }
836
+ await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
837
+ return;
838
+ }
839
+ ex.audit.running = true;
840
+ ex.audit.rounds += 1;
841
+ updateStatusWidget(ctx);
842
+ let outcome = null as Awaited<ReturnType<typeof runCompletionAudit>>;
843
+ try {
844
+ outcome = auditRunnerForTests
845
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: ex.audit.rounds })
846
+ : await runCompletionAudit(ctx, {
847
+ planPath: ex.planPath,
848
+ checklist: ex.items,
849
+ tasks: ex.tasks,
850
+ round: ex.audit.rounds,
851
+ signal: ctx.signal,
852
+ });
853
+ } finally {
854
+ ex.audit.running = false;
855
+ }
856
+ // Fail-closed: checks the outcome did not affirmatively pass are failed.
857
+ const reportedPass = new Set(outcome?.passed ?? []);
858
+ const failed = pendingChecks
859
+ .map((item) => item.id)
860
+ .filter((id) => !reportedPass.has(id));
861
+ if (failed.length === 0) {
862
+ ex.audit.failed = [];
863
+ withExecutionCheckpoint(ctx, (cp) =>
864
+ applyExecutionProgress(cp, {
865
+ tasks: taskProgressMap(ex.tasks),
866
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
867
+ audit: { rounds: ex.audit.rounds, passed: true },
868
+ }),
869
+ );
870
+ await completeExecution(ctx);
871
+ return;
872
+ }
873
+ // Failed checks: roll their covered tasks back (always — an unreported or
874
+ // infra-failed audit must reopen work so the loop can continue) and
875
+ // persist progress including checks that passed earlier rounds.
876
+ for (const id of failed) {
877
+ const item = ex.items.find((candidate) => candidate.id === id);
878
+ if (item) item.done = false;
879
+ }
880
+ ex.audit.failed = failed;
881
+ const rolledBack: string[] = [];
882
+ for (const id of failed) {
883
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
884
+ }
885
+ withExecutionCheckpoint(ctx, (cp) =>
886
+ applyExecutionProgress(cp, {
887
+ tasks: taskProgressMap(ex.tasks),
888
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
889
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
890
+ }),
891
+ );
892
+ ex.stall.rounds = 0;
893
+ ex.stall.lastSnapshot = stallSnapshot();
894
+ persist(ctx);
895
+ updateStatusWidget(ctx);
896
+ const report = outcome?.report ?? "(audit subagent failed to run)";
897
+ // v0.7.1: the failure notification now wakes the agent so it can fix the
898
+ // rolled-back work without the user having to poke the run (root cause of
899
+ // the observed stall). The wake is the ONLY turn this settle emits — the
900
+ // rollback below also marks the runtime handled so the continuation path
901
+ // cannot add a second EXECUTION_CONTINUE wake for the same settle (Q-2).
902
+ const stranded = rolledBack.length === 0;
903
+ messaging().sendMessage(
904
+ {
905
+ customType: "pi-plans-audit-failed",
906
+ content: `**pi-plans: completion audit round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the audit reruns automatically once all tasks are terminal again.${ex.audit.rounds >= AUDIT_MAX_ROUNDS ? ` This was round ${AUDIT_MAX_ROUNDS} of ${AUDIT_MAX_ROUNDS}: interactive sessions pause for review; the next terminal-task cycle stops or pauses the run.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the audit automatically.` : ""}\n\n---\n${report.slice(0, 4000)}`,
907
+ display: true,
908
+ },
909
+ { triggerTurn: true },
910
+ );
911
+ // Q-2 (single wake): this message owns the settle's only turn. Mark the
912
+ // runtime handled so maybeContinuationFollowUp stays silent.
913
+ const runtime = currentContinuationRuntime(ctx);
914
+ if (runtime) runtime.handled = true;
915
+ }
916
+
917
+ /** True when the session can surface a pause to a human (D-022): interactive
918
+ * TUI/RPC sessions that are not running under PI_PLANS_AUTO_APPROVE. */
919
+ function isInteractiveSession(ctx: ExtensionContext): boolean {
920
+ if ((ctx.mode !== "tui" && ctx.mode !== "rpc") || ctx.hasUI !== true) return false;
921
+ return !isAutoApproveEnabledLocal();
922
+ }
923
+
835
924
  const EXECUTION_RESUME_CUSTOM_TYPE = "pi-plans-exec-resume";
836
925
 
837
926
  function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase): { settings: PiPlansVccSettings; runId: string; artifactDir: string } | null {
@@ -848,12 +937,13 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
848
937
  }
849
938
 
850
939
  function executionVccContext(): PiPlansVccPhaseContext {
940
+ const current = execution ? currentTask(execution.tasks) : null;
851
941
  return {
852
942
  phase: "execution",
853
943
  planPath: execution?.planPath ?? null,
854
- currentI: execution?.currentI ?? null,
944
+ currentI: current?.id ?? null,
855
945
  remainingVerifierIds: execution?.items.filter((item) => !item.done).map((item) => item.id) ?? [],
856
- implementationIds: execution?.implItems?.map((item) => item.id) ?? [],
946
+ implementationIds: execution ? flattenTaskViews(execution.tasks).map((task) => task.id) : [],
857
947
  };
858
948
  }
859
949
 
@@ -897,7 +987,6 @@ export function buildExecutionCompactionResult(event: SessionBeforeCompactEvent,
897
987
  }
898
988
 
899
989
  export function handleExecutionBeforeCompact(
900
- pi: ExtensionAPI,
901
990
  ctx: ExtensionContext,
902
991
  event: SessionBeforeCompactEvent,
903
992
  ): SessionBeforeCompactResult | undefined {
@@ -928,7 +1017,7 @@ export function handleExecutionBeforeCompact(
928
1017
  state.pendingStats = built.stats;
929
1018
  state.pendingFollowUpPrompt = built.followUpPrompt;
930
1019
  state.pendingContinueAfterThresholdCompact = built.settings.continueAfterThresholdCompact;
931
- requestExecutionFlush(pi, ctx);
1020
+ requestExecutionFlush();
932
1021
  return { compaction: built.compaction };
933
1022
  }
934
1023
 
@@ -936,7 +1025,7 @@ function runtimePiVersion(ctx: ExtensionContext): unknown {
936
1025
  return (ctx as ExtensionContext & { piVersion?: unknown }).piVersion ?? VERSION;
937
1026
  }
938
1027
 
939
- export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1028
+ export async function handleExecutionCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
940
1029
  if (!execution) return;
941
1030
  const state = ensureExecutionCompactionState(ctx);
942
1031
  const stats = state.pendingStats;
@@ -956,10 +1045,10 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
956
1045
  if (!event.willRetry && stats) {
957
1046
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
958
1047
  if (followUpPrompt) {
959
- await pi.sendUserMessage?.(followUpPrompt);
1048
+ await messaging().sendUserMessage(followUpPrompt);
960
1049
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
961
1050
  state.resumeGuard = true;
962
- pi.sendMessage(
1051
+ messaging().sendMessage(
963
1052
  {
964
1053
  customType: EXECUTION_RESUME_CUSTOM_TYPE,
965
1054
  content: EXECUTION_COMPACTION_RESUME_MESSAGE,
@@ -969,18 +1058,15 @@ export async function handleExecutionCompact(pi: ExtensionAPI, ctx: ExtensionCon
969
1058
  );
970
1059
  }
971
1060
  }
972
- requestExecutionFlush(pi, ctx);
1061
+ requestExecutionFlush();
973
1062
  updateStatusWidget(ctx);
974
1063
  }
975
1064
 
976
-
977
- export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1065
+ export function handleExecutionCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
978
1066
  if (!execution) return;
979
1067
  const state = executionCompactionState(ctx);
980
1068
  const terminal = isTerminalCompactionFailure(event);
981
1069
  if (terminal) {
982
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
983
- // only after real growth or high-watermark pressure so the loop stops.
984
1070
  if (state) {
985
1071
  state.inFlight = false;
986
1072
  state.resumeGuard = false;
@@ -997,7 +1083,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
997
1083
  ? "pi-plans: compaction found nothing to summarize; backing off until the session grows past the keep-recent window."
998
1084
  : "pi-plans: compaction was aborted (provider interruption, user cancel, or a competing manual compact); backing off until the session grows or usage nears the window.";
999
1085
  ctx.ui.notify(message, "info");
1000
- requestExecutionFlush(pi, ctx);
1086
+ requestExecutionFlush();
1001
1087
  return;
1002
1088
  }
1003
1089
  if (state) {
@@ -1014,7 +1100,7 @@ export function handleExecutionCompactFailed(pi: ExtensionAPI, ctx: ExtensionCon
1014
1100
  `pi-plans: compaction failed (${event.reason}); execution remains active and will wait for the next eligible turn.`,
1015
1101
  "warning",
1016
1102
  );
1017
- requestExecutionFlush(pi, ctx);
1103
+ requestExecutionFlush();
1018
1104
  }
1019
1105
 
1020
1106
  export function filterExecutionResumeMessages<T extends { customType?: string }>(messages: T[]): T[] {
@@ -1024,24 +1110,12 @@ export function filterExecutionResumeMessages<T extends { customType?: string }>
1024
1110
  // ---------------------------------------------------------------------------
1025
1111
  // Planning-phase compaction: Pi core owns scheduling; this hook customizes
1026
1112
  // active planning compact events with the same VCC builder used by execution.
1027
- // The two state machines are kept independent (different memory slot and
1028
- // snapshot key) so execution never bleeds into planning.
1029
1113
  // ---------------------------------------------------------------------------
1030
1114
 
1031
1115
  export const PLANNING_RUN_START_CUSTOM_TYPE = "pi-plans-run-start";
1032
1116
  export const PLANNING_PLAN_WRITTEN_CUSTOM_TYPE = "pi-plans-plan-written";
1033
1117
  const PLANNING_RESUME_CUSTOM_TYPE = "pi-plans-plan-resume";
1034
1118
 
1035
- // ---------------------------------------------------------------------------
1036
- // Pre-plan compaction: right after `plans start-run` creates a new planning
1037
- // run, the extension triggers one VCC compaction so the new plan starts on a
1038
- // lean context (LLM reasoning degrades with longer context; see PLAN
1039
- // preplan-compact). The pending flag is session-scoped and opportunistic: it
1040
- // is set by the start-run tool case and consumed by the plans tool_result
1041
- // hook in index.ts, which requests the extension-context compact action and
1042
- // resumes planning exactly once regardless of success or failure.
1043
- // ---------------------------------------------------------------------------
1044
-
1045
1119
  export { PLANNING_PREPLAN_COMPACT_HINT };
1046
1120
  export const PLANNING_PREPLAN_RESUME_CUSTOM_TYPE = "pi-plans-preplan-resume";
1047
1121
 
@@ -1061,11 +1135,8 @@ export function consumePrePlanCompactPending(ctx: ExtensionContext): PrePlanComp
1061
1135
  return pending;
1062
1136
  }
1063
1137
 
1064
- /** Hidden resume message after the pre-plan compaction settles (success or
1065
- * failure): Pi's manual compaction never continues the aborted turn, so the
1066
- * planning workflow is continued exactly once from here. */
1067
- export function sendPrePlanCompactResume(pi: ExtensionAPI): void {
1068
- pi.sendMessage?.(
1138
+ export function sendPrePlanCompactResume(ctx: ExtensionContext): void {
1139
+ messaging().sendMessage(
1069
1140
  {
1070
1141
  customType: PLANNING_PREPLAN_RESUME_CUSTOM_TYPE,
1071
1142
  content: "Continue planning.",
@@ -1082,7 +1153,6 @@ interface PlanningCompactionState {
1082
1153
  lastAttemptReason: "manual" | "threshold" | "overflow" | null;
1083
1154
  lastSuccessfulUsagePercent: number | null;
1084
1155
  lastSuccessfulAt: string | null;
1085
- /** Terminal "nothing to compact" backoff: tokens observed when Pi refused. */
1086
1156
  terminalBackoffTokens: number | null;
1087
1157
  pendingStats: VccCompactionStats | null;
1088
1158
  pendingFollowUpPrompt: string | null;
@@ -1110,8 +1180,6 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1110
1180
  if (message.includes("nothing to compact") || message.includes("already compacted") || message.includes("session too small")) {
1111
1181
  return { kind: "content" };
1112
1182
  }
1113
- // abort/stream class: explicit event names only, so that provider blips
1114
- // (network down, etc.) stay retryable.
1115
1183
  const abortPatterns = [
1116
1184
  "this operation was aborted",
1117
1185
  "aborted",
@@ -1123,20 +1191,12 @@ function isTerminalCompactionFailure(event: { errorMessage?: string; aborted?: b
1123
1191
  if (abortPatterns.some((pattern) => message.includes(pattern))) {
1124
1192
  return { kind: "abort-stream" };
1125
1193
  }
1126
- // Aborted with no recognized message: still an abort-class terminal so the
1127
- // next eligible turn does not immediately retry the same operation.
1128
1194
  if (event.aborted === true) {
1129
1195
  return { kind: "abort-stream" };
1130
1196
  }
1131
1197
  return null;
1132
1198
  }
1133
1199
 
1134
- /** Session-scoped phase-local "compaction in flight" guard.
1135
- * - Set on `session_before_compact` for the phase attributed by the custom
1136
- * instructions hint; auto-compaction (no hint) marks both phases defensively.
1137
- * - Cleared on `session_compact` and `session_compact_failed`.
1138
- * - Retained so lifecycle events expose the same phase-local state to tests
1139
- * and future Pi core schema additions. */
1140
1200
  type CompactionPhase = "planning" | "execution";
1141
1201
 
1142
1202
  function compactionLifecycleStore(ctx: ExtensionContext): {
@@ -1165,18 +1225,11 @@ export function noteCompactionStarted(ctx: ExtensionContext, customInstructions:
1165
1225
  } else if (isExecutionCustomInstructions(customInstructions)) {
1166
1226
  store.execution = true;
1167
1227
  } else {
1168
- // Auto-compaction (threshold/overflow/manual without our hint) marks both.
1169
1228
  store.planning = true;
1170
1229
  store.execution = true;
1171
1230
  }
1172
1231
  }
1173
1232
 
1174
- /** Pi core's `SessionCompactEvent` / `SessionCompactFailedEvent` do not carry
1175
- * `customInstructions` in any emission site, so the END side has no way to
1176
- * know which phase the compaction belonged to. Clearing both phases is the
1177
- * safe default — the per-phase start side (above) already encodes the hint
1178
- * attribution. The hint parameter is retained for API symmetry and future
1179
- * Pi core schema additions. */
1180
1233
  export function noteCompactionEnded(ctx: ExtensionContext, _customInstructions: unknown): void {
1181
1234
  const store = compactionLifecycleStore(ctx);
1182
1235
  store.planning = false;
@@ -1200,15 +1253,11 @@ export function consumePlanningCompactionResumeGuard(ctx: ExtensionContext): boo
1200
1253
  }
1201
1254
 
1202
1255
  export function refreshPlanningCompactionCooldown(_ctx: ExtensionContext): void {
1203
- // Pi core owns scheduling; retained for lifecycle compatibility only.
1256
+ // Retained for lifecycle compatibility only.
1204
1257
  }
1205
1258
 
1206
1259
  export function requestPlanningCompaction(_ctx: ExtensionContext): void {
1207
- // Generic proactive pi-plans compaction is intentionally disabled. Manual,
1208
- // threshold, and overflow compactions are handled by session_before_compact.
1209
- // The single exception is the pre-plan compaction: index.ts requests the
1210
- // extension-context compact action from the plans tool_result hook right
1211
- // after start-run (see PLANNING_PREPLAN_COMPACT_HINT).
1260
+ // Manual, threshold, and overflow compactions are handled by session_before_compact.
1212
1261
  }
1213
1262
 
1214
1263
  function buildPlanningVccResult(event: SessionBeforeCompactEvent, ctx: ExtensionContext): VccCompactionBuildResult | null {
@@ -1233,7 +1282,6 @@ export function buildPlanningCompactionResult(event: SessionBeforeCompactEvent,
1233
1282
  }
1234
1283
 
1235
1284
  export function handlePlanningBeforeCompact(
1236
- pi: ExtensionAPI,
1237
1285
  ctx: ExtensionContext,
1238
1286
  event: SessionBeforeCompactEvent,
1239
1287
  ): SessionBeforeCompactResult | undefined {
@@ -1267,7 +1315,7 @@ export function handlePlanningBeforeCompact(
1267
1315
  return { compaction: built.compaction };
1268
1316
  }
1269
1317
 
1270
- export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1318
+ export async function handlePlanningCompact(ctx: ExtensionContext, event: SessionCompactEvent): Promise<void> {
1271
1319
  if (getExecution()) return;
1272
1320
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1273
1321
  const state = session.__planningCompaction;
@@ -1288,10 +1336,10 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1288
1336
  if (!event.willRetry && stats) {
1289
1337
  ctx.ui.notify(formatVccCompactionStats(stats), "info");
1290
1338
  if (followUpPrompt) {
1291
- await pi.sendUserMessage?.(followUpPrompt);
1339
+ await messaging().sendUserMessage(followUpPrompt);
1292
1340
  } else if ((event.reason === "threshold" || event.reason === "overflow") && shouldScheduleAutoContinue(continueAfterThresholdCompact, runtimePiVersion(ctx))) {
1293
1341
  state.resumeGuard = true;
1294
- pi.sendMessage(
1342
+ messaging().sendMessage(
1295
1343
  {
1296
1344
  customType: PLANNING_RESUME_CUSTOM_TYPE,
1297
1345
  content: "Continue planning.",
@@ -1303,16 +1351,13 @@ export async function handlePlanningCompact(pi: ExtensionAPI, ctx: ExtensionCont
1303
1351
  }
1304
1352
  }
1305
1353
 
1306
-
1307
- export function handlePlanningCompactFailed(pi: ExtensionAPI, ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1354
+ export function handlePlanningCompactFailed(ctx: ExtensionContext, event: SessionCompactFailedEvent): void {
1308
1355
  if (getExecution()) return;
1309
1356
  const session = ctx.sessionManager as unknown as { __planningCompaction?: PlanningCompactionState };
1310
1357
  const state = session.__planningCompaction;
1311
1358
  if (!state) return;
1312
1359
  const terminal = isTerminalCompactionFailure(event);
1313
1360
  if (terminal) {
1314
- // Pi refused or aborted the compaction. Hold the cooldown and re-arm
1315
- // only after real growth or high-watermark pressure so the loop stops.
1316
1361
  state.inFlight = false;
1317
1362
  state.resumeGuard = false;
1318
1363
  state.cooldownActive = true;
@@ -1345,19 +1390,17 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
1345
1390
  return messages.filter((message) => message.customType !== PLANNING_RESUME_CUSTOM_TYPE && message.customType !== PLANNING_PREPLAN_RESUME_CUSTOM_TYPE);
1346
1391
  }
1347
1392
 
1348
- export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): Promise<void> {
1393
+ export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
1349
1394
  if (!execution) return;
1350
1395
  resetExecutionCompactionState(ctx);
1351
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1352
1396
  pendingExecutionFlush = false;
1353
- persist(pi);
1354
- // Checkpoint first: withExecutionCheckpoint guards on the live execution.
1397
+ persist(ctx);
1355
1398
  withExecutionCheckpoint(ctx, (cp) => applyExecutionStopped(cp, reason));
1356
1399
  execution = null;
1357
1400
  executionRunId = null;
1358
- goalWaitRuntime = null;
1359
- pi.appendEntry("pi-plans-exec-cleared", { reason });
1360
- pi.sendMessage(
1401
+ continuationRuntime = null;
1402
+ messaging().appendEntry("pi-plans-exec-cleared", { reason });
1403
+ messaging().sendMessage(
1361
1404
  {
1362
1405
  customType: "pi-plans-exec-stop",
1363
1406
  content: `**pi-plans: execution stopped** — ${reason}`,
@@ -1376,84 +1419,76 @@ export async function stopExecution(pi: ExtensionAPI, ctx: ExtensionContext, rea
1376
1419
  updateStatusWidget(ctx);
1377
1420
  }
1378
1421
 
1379
- /** Apply [DONE:VC-xxx] markers from an assistant message. Returns changed ids. */
1380
- export function applyDoneMarkers(text: string): string[] {
1381
- if (!execution) return [];
1382
- const changed: string[] = [];
1383
- for (const id of scanDoneMarkers(text)) {
1384
- const item = execution.items.find((candidate) => candidate.id === id && !candidate.done);
1385
- if (item) {
1386
- item.done = true;
1387
- changed.push(id);
1388
- }
1389
- }
1390
- return changed;
1422
+ /** v0.7.1: the per-settle latch, created on demand so no execution-construction
1423
+ * path can leave it undefined (a missing latch must degrade to "no latch",
1424
+ * never throw inside a lifecycle handler). */
1425
+ function auditLatchOf(ex: ExecState): ExecState["auditLatch"] {
1426
+ if (!ex.auditLatch) ex.auditLatch = { auditedThisSettle: false, activity: 0 };
1427
+ return ex.auditLatch;
1391
1428
  }
1392
1429
 
1393
- /**
1394
- * Apply [I-xxx:implemented|validating] markers from an assistant message.
1395
- * Unknown I-ids are silently ignored; later markers overwrite earlier ones.
1396
- * Returns the ids whose state actually changed.
1397
- */
1398
- export function applyImplMarkers(text: string): string[] {
1399
- if (!execution?.implItems?.length) return [];
1400
- const known = new Set(execution.implItems.map((impl) => impl.id));
1401
- execution.implStatus ??= {};
1402
- const changed: string[] = [];
1403
- for (const marker of scanImplMarkers(text)) {
1404
- if (!known.has(marker.id)) continue;
1405
- const previous = execution.implStatus[marker.id];
1406
- execution.implStatus[marker.id] = marker.state;
1407
- if (previous !== marker.state) changed.push(marker.id);
1408
- }
1409
- return changed;
1430
+ function stallSnapshot(): string {
1431
+ if (!execution) return "";
1432
+ // v0.7.1: the snapshot carries the round's tool-activity counter, so a round
1433
+ // in which the agent legitimately did work (read code, run commands) counts
1434
+ // as progress even when no task changed status. Only a round with neither a
1435
+ // status change NOR a successful tool result is "no progress".
1436
+ return JSON.stringify({ tasks: taskProgressMap(execution.tasks), activity: auditLatchOf(execution).activity });
1410
1437
  }
1411
1438
 
1412
- export function applyCurrentIMarker(text: string): boolean {
1413
- if (!execution?.implItems?.length) return false;
1414
- const markers = scanCurrentIMarkers(text);
1415
- const resolved = resolveCurrentI(execution.implItems, markers, execution.currentI);
1416
- if (!resolved || resolved === execution.currentI) return false;
1417
- execution.currentI = resolved;
1418
- return true;
1439
+ /**
1440
+ * v0.7.1: shared "the completion audit is owed" predicate. Every entry point
1441
+ * that can start the audit (turn_end, agent_before_settle, restoreFromSession,
1442
+ * the resume path) goes through this so they can never disagree.
1443
+ *
1444
+ * Semantics (unchanged from restoreFromSession's guard, F-006): a check that
1445
+ * covers no task never gates completion, so only auditable checks count.
1446
+ */
1447
+ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1448
+ if (!ex) return false;
1449
+ // A paused run is never self-driven: the stall / audit-cap pause is an
1450
+ // explicit "hand control back" signal, and resuming it is the user's call.
1451
+ // This also bounds the zero-input continue loop in agent_before_settle.
1452
+ if (ex.stall.paused) return false;
1453
+ if (ex.audit.running) return false;
1454
+ return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
1419
1455
  }
1420
1456
 
1421
- export function isExecutionComplete(): boolean {
1422
- return execution !== null && execution.items.length > 0 && execution.items.every((item) => item.done);
1457
+ /** v0.7.1: record that this settle already ran (or declined) its audit, so a
1458
+ * second entry point in the same settle cannot re-consume a round. */
1459
+ function latchAuditThisSettle(): void {
1460
+ if (execution) auditLatchOf(execution).auditedThisSettle = true;
1423
1461
  }
1424
1462
 
1425
- function goalWaitSnapshot(): string {
1426
- if (!execution) return "";
1427
- return JSON.stringify({
1428
- done: execution.items
1429
- .filter((item) => item.done)
1430
- .map((item) => item.id)
1431
- .sort()
1432
- .join("|"),
1433
- implStatus: execution.implStatus ?? {},
1434
- currentI: execution.currentI ?? null,
1435
- });
1463
+ /** v0.7.1: called on agent_start — a new agent run is a new settle window, so
1464
+ * the latch and the activity counter both reset here. */
1465
+ function resetSettleLatch(): void {
1466
+ if (!execution) return;
1467
+ const latch = auditLatchOf(execution);
1468
+ latch.auditedThisSettle = false;
1469
+ latch.activity = 0;
1436
1470
  }
1437
1471
 
1438
- function pauseGoalWait(pi: ExtensionAPI, ctx: ExtensionContext, reason: string): void {
1472
+ function pauseForStall(ctx: ExtensionContext, reason: string): void {
1439
1473
  const ex = getExecution();
1440
- if (!ex?.goalWait) return;
1441
- ex.goalWait.paused = true;
1442
- ex.goalWait.pausedReason = reason;
1443
- persist(pi);
1474
+ if (!ex) return;
1475
+ ex.stall.paused = true;
1476
+ ex.stall.pausedReason = reason;
1477
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: reason }));
1478
+ persist(ctx);
1444
1479
  ctx.ui.notify?.(
1445
- `pi-plans: goal-wait paused (${reason}). Send any message or run /plans-execute to resume.`,
1480
+ `pi-plans: execution paused (${reason}). Send any message or run /plans-execute to resume.`,
1446
1481
  "warning",
1447
1482
  );
1448
1483
  updateStatusWidget(ctx);
1449
1484
  }
1450
1485
 
1451
- function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1486
+ function canWakeExecution(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1452
1487
  const compaction = executionCompactionState(ctx);
1453
- return currentGoalWaitRuntime(ctx) === runtime
1488
+ return currentContinuationRuntime(ctx) === runtime
1454
1489
  && (ctx.mode === "tui" || ctx.mode === "rpc")
1455
- && !isExecutionComplete()
1456
- && !runtime.owner.goalWait?.paused
1490
+ && !allTasksTerminal(runtime.owner.tasks)
1491
+ && !runtime.owner.stall.paused
1457
1492
  && ctx.isIdle()
1458
1493
  && !ctx.hasPendingMessages()
1459
1494
  && !ctx.signal?.aborted
@@ -1463,15 +1498,14 @@ function canWakeExecution(ctx: ExtensionContext, runtime: GoalWaitRuntime): bool
1463
1498
  && compaction?.pendingFollowUpPrompt == null;
1464
1499
  }
1465
1500
 
1466
- function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: GoalWaitRuntime): boolean {
1501
+ function sendContinuationWake(ctx: ExtensionContext, runtime: ContinuationRuntime): boolean {
1467
1502
  if (!canWakeExecution(ctx, runtime)) return false;
1468
1503
  try {
1469
- // Custom messages bypass before_agent_start, so carry fresh execution rules.
1470
1504
  const content = executionContextMessage(ctx);
1471
1505
  if (!content) return false;
1472
1506
  runtime.wakeId = randomUUID();
1473
- pi.sendMessage({
1474
- customType: GOAL_WAIT_CUSTOM_TYPE,
1507
+ messaging().sendMessage({
1508
+ customType: EXECUTION_CONTINUE_CUSTOM_TYPE,
1475
1509
  content,
1476
1510
  display: false,
1477
1511
  details: { wakeId: runtime.wakeId },
@@ -1479,120 +1513,133 @@ function sendGoalWaitWake(pi: ExtensionAPI, ctx: ExtensionContext, runtime: Goal
1479
1513
  return true;
1480
1514
  } catch (error) {
1481
1515
  runtime.wakeId = undefined;
1482
- pauseGoalWait(pi, ctx, `continuation failed: ${String(error)}`);
1516
+ pauseForStall(ctx, `continuation failed: ${String(error)}`);
1483
1517
  return false;
1484
1518
  }
1485
1519
  }
1486
1520
 
1487
1521
  /** Only a fully settled agent run can need an extra wake, never a tool turn. */
1488
- function maybeGoalWaitFollowUp(pi: ExtensionAPI, ctx: ExtensionContext): void {
1489
- const runtime = currentGoalWaitRuntime(ctx);
1522
+ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
1523
+ const runtime = currentContinuationRuntime(ctx);
1490
1524
  if (!runtime || runtime.handled || !ctx.isIdle()) return;
1491
1525
  if (ctx.mode !== "tui" && ctx.mode !== "rpc") return;
1492
1526
  if (runtime.stopReason === "error" || runtime.stopReason === "aborted" || ctx.signal?.aborted) {
1493
1527
  runtime.handled = true;
1494
- pauseGoalWait(pi, ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1528
+ pauseForStall(ctx, runtime.stopReason === "error" ? "agent failed" : "agent interrupted");
1495
1529
  return;
1496
1530
  }
1531
+ // v0.7.1: a fully-terminal run that still owes an audit is NOT a
1532
+ // continuation case — the audit owns that state (agent_before_settle).
1533
+ // Return without waking: emitting EXECUTION_CONTINUE here would send the
1534
+ // agent back to redo work it has already finished (F-003).
1535
+ if (execution && allTasksTerminal(execution.tasks) && pendingAudit(execution)) return;
1497
1536
  if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
1498
1537
  runtime.handled = true;
1499
1538
  const ex = runtime.owner;
1500
- ex.goalWait ??= { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false };
1501
- const goalWait = ex.goalWait;
1502
- const snapshot = goalWaitSnapshot();
1503
- const changed = goalWait.lastMarkers !== null && snapshot !== goalWait.lastMarkers;
1504
- goalWait.lastMarkers = snapshot;
1539
+ const snapshot = stallSnapshot();
1540
+ const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
1541
+ ex.stall.lastSnapshot = snapshot;
1505
1542
  if (changed) {
1506
- goalWait.noProgressRounds = 0;
1507
- goalWait.waitRounds = 0;
1508
- } else if (/waiting for/i.test(runtime.text)) {
1509
- goalWait.waitRounds += 1;
1543
+ ex.stall.rounds = 0;
1510
1544
  } else {
1511
- goalWait.noProgressRounds += 1;
1512
- }
1513
- if (goalWait.noProgressRounds >= GOAL_WAIT_MAX_NO_PROGRESS) {
1514
- pauseGoalWait(pi, ctx, `no progress in ${goalWait.noProgressRounds} rounds`);
1515
- return;
1545
+ ex.stall.rounds += 1;
1516
1546
  }
1517
- if (goalWait.waitRounds >= GOAL_WAIT_MAX_WAITING) {
1518
- pauseGoalWait(pi, ctx, `waiting without progress for ${goalWait.waitRounds} rounds`);
1547
+ if (ex.stall.rounds >= STALL_MAX_ROUNDS) {
1548
+ pauseForStall(ctx, `no task-status change in ${ex.stall.rounds} rounds`);
1519
1549
  return;
1520
1550
  }
1521
- persist(pi);
1551
+ persist(ctx);
1522
1552
  updateStatusWidget(ctx);
1523
- // No await between the live gate and dispatch: another input cannot interleave.
1524
- sendGoalWaitWake(pi, ctx, runtime);
1553
+ sendContinuationWake(ctx, runtime);
1525
1554
  }
1526
1555
 
1527
1556
  export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
1528
- return messages.filter((message) => message.customType !== GOAL_WAIT_CUSTOM_TYPE
1529
- || (goalWaitRuntime?.owner === execution && goalWaitRuntime?.wakeId !== undefined
1530
- && (message.details as { wakeId?: unknown } | undefined)?.wakeId === goalWaitRuntime.wakeId));
1557
+ // v0.6.1: continuation wakes are one-shot; stale ones (including the
1558
+ // legacy v0.6.0 goal-wait type) never replay after a restart.
1559
+ return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
1560
+ && message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
1531
1561
  }
1532
1562
 
1533
- /** Called only for genuine user input or an explicit same-execution resume. */
1534
- export function resumeGoalWaitIfPaused(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
1563
+ export function filterContinuationMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
1564
+ return filterGoalWaitMessages(messages);
1565
+ }
1566
+
1567
+ /** Prefix of the stall reason used for the audit-cap pause (D-022). */
1568
+ const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
1569
+
1570
+ /** Called for genuine user input or an explicit same-execution resume.
1571
+ * Resuming an audit-cap pause grants a fresh audit budget (three more
1572
+ * rounds): the user's explicit resume IS the decision to keep auditing —
1573
+ * without this reset the cap pause could never be lifted productively. */
1574
+ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
1535
1575
  const ex = getExecution();
1536
- if (!ex?.goalWait?.paused || !currentGoalWaitRuntime(ctx)) return false;
1537
- ex.goalWait.paused = false;
1538
- ex.goalWait.pausedReason = undefined;
1539
- ex.goalWait.noProgressRounds = 0;
1540
- ex.goalWait.waitRounds = 0;
1541
- ex.goalWait.lastMarkers = goalWaitSnapshot();
1542
- persist(pi);
1576
+ if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
1577
+ const wasAuditCap = (ex.stall.pausedReason ?? "").startsWith(AUDIT_CAP_PAUSE_PREFIX);
1578
+ ex.stall.paused = false;
1579
+ ex.stall.pausedReason = undefined;
1580
+ ex.stall.rounds = 0;
1581
+ ex.stall.lastSnapshot = stallSnapshot();
1582
+ if (wasAuditCap) {
1583
+ ex.audit.rounds = 0;
1584
+ ex.audit.failed = [];
1585
+ withExecutionCheckpoint(ctx, (cp) =>
1586
+ applyExecutionProgress(cp, {
1587
+ tasks: taskProgressMap(ex.tasks),
1588
+ audit: { rounds: 0, lastResult: undefined },
1589
+ pausedReason: null,
1590
+ }),
1591
+ );
1592
+ } else {
1593
+ withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
1594
+ }
1595
+ persist(ctx);
1543
1596
  updateStatusWidget(ctx);
1544
1597
  return true;
1545
1598
  }
1546
1599
 
1547
- export function resumeActiveExecution(pi: ExtensionAPI, ctx: ExtensionContext): boolean {
1548
- if (!resumeGoalWaitIfPaused(pi, ctx)) return false;
1549
- const runtime = currentGoalWaitRuntime(ctx)!;
1600
+ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
1601
+ // v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
1602
+ // to `return false` here, so `/plans-execute` answered "already executing"
1603
+ // and the run stayed stranded until a full re-entry or a session restore.
1604
+ // It is not paused, so the pause path below cannot see it — check it first
1605
+ // and run the owed audit instead of reporting "nothing to resume".
1606
+ if (!execution?.stall.paused && pendingAudit()) {
1607
+ latchAuditThisSettle();
1608
+ void runAuditFlow(ctx).catch(() => { /* surfaced via the audit message */ });
1609
+ return true;
1610
+ }
1611
+ if (!resumeGoalWaitIfPaused(ctx)) return false;
1612
+ const runtime = currentContinuationRuntime(ctx)!;
1550
1613
  if (canWakeExecution(ctx, runtime)) {
1551
1614
  runtime.handled = true;
1552
- sendGoalWaitWake(pi, ctx, runtime);
1615
+ sendContinuationWake(ctx, runtime);
1553
1616
  }
1554
1617
  return true;
1555
1618
  }
1556
1619
 
1557
- export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext): Promise<void> {
1620
+ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
1558
1621
  if (!execution) return;
1559
1622
  resetExecutionCompactionState(ctx);
1560
- // Final synchronous write: drain any deferred flush and land the last snapshot.
1561
1623
  pendingExecutionFlush = false;
1562
- persist(pi);
1563
-
1564
- const summary = execution.items.map((item) => `- ✅ \`${item.id}\` ${item.text.split(";")[0]}`).join("\n");
1624
+ persist(ctx);
1625
+ const flat = flattenTaskViews(execution.tasks);
1626
+ const summary = flat
1627
+ .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
1628
+ .join("\n");
1565
1629
  const planPath = execution.planPath;
1566
- // Checkpoint first (live-execution guard), then clear the session state.
1567
1630
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
1568
1631
  execution = null;
1569
1632
  executionRunId = null;
1570
- goalWaitRuntime = null;
1571
- pi.appendEntry("pi-plans-exec-cleared", { reason: "complete" });
1572
- // Post-execution goal-running continuation: in interactive sessions, attach
1573
- // the continuation block and trigger a new turn so the agent immediately
1574
- // enters the implementation-review loop. Headless sessions keep the silent
1575
- // completion behavior. Both completeExecution call sites (turn_end and the
1576
- // restoreFromSession recovery path) share this behavior.
1577
- const interactive = ctx.hasUI === true;
1578
- // Skill-aware continuation: the reviewer-count default follows the active
1579
- // run's skill (D-1/D-4), so the prompt names the run's own recommended
1580
- // count instead of a static guess.
1581
- const activeRunForPrompt = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1582
- const content = interactive
1583
- ? `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}\n\n${ameliorationPromptText(activeRunForPrompt?.skill)}`
1584
- : `**Plan complete!** ✅ \`${planPath}\`\n\n${summary}`;
1585
- pi.sendMessage(
1633
+ continuationRuntime = null;
1634
+ messaging().appendEntry("pi-plans-exec-cleared", { reason: "complete" });
1635
+ messaging().sendMessage(
1586
1636
  {
1587
1637
  customType: "pi-plans-complete",
1588
- content,
1638
+ content: `**Plan complete!** ✅ \`${planPath}\` — completion audit passed.\n\n${summary}`,
1589
1639
  display: true,
1590
1640
  },
1591
- { triggerTurn: interactive },
1641
+ { triggerTurn: false },
1592
1642
  );
1593
- if (interactive) {
1594
- pi.appendEntry("pi-plans-ameliorate", { planPath, phase: "goal-started", rounds: null, currentRound: 0 });
1595
- }
1596
1643
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1597
1644
  if (active) {
1598
1645
  try {
@@ -1604,55 +1651,45 @@ export async function completeExecution(pi: ExtensionAPI, ctx: ExtensionContext)
1604
1651
  updateStatusWidget(ctx);
1605
1652
  }
1606
1653
 
1607
- /** Instructions appended to the post-execution completion message in
1608
- * interactive sessions, telling the agent to enter the goal-running
1609
- * implementation-review loop. Skill-aware: the reviewer-count question's
1610
- * recommended option follows the run's skill (plan-big / plan-with-refs → 3,
1611
- * others → 1). Termination options are single-sourced from
1612
- * src/termination-prompt.ts (shared with the ask_choice trailing branch). */
1613
- export function ameliorationPromptText(skill: string | undefined): string {
1614
- return `---
1615
- Goal-running continuation: immediately ask the user now via ask_choice (autoComplete: false, in the session language) the termination question: "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(skill)} Then keep running the implementation-review loop without asking whether to continue; the goal-wait option keeps the loop running until no unpassed VCs remain.`;
1616
- }
1617
-
1618
1654
  /** Injection text for before_agent_start while executing. */
1619
1655
  export function executionContextMessage(ctx: ExtensionContext): string | null {
1620
1656
  if (!execution) return null;
1621
- const remaining = execution.items.filter((item) => !item.done);
1622
- const list =
1623
- remaining.map((item) => `- \`${item.id}\` ${item.text}`).join("\n") || "(none — report completion now)";
1624
- // Live read: the injected guidance and the tool wrappers share the same
1625
- // tri-state, so they can never contradict each other mid-run.
1657
+ const flat = flattenTaskViews(execution.tasks);
1658
+ const open = flat.filter((task) => !taskIsTerminal(task));
1659
+ const cur = currentTask(execution.tasks);
1660
+ const currentWave = cur?.wave ?? 1;
1661
+ const inWave = open.filter((task) => task.wave === currentWave);
1662
+ const waveList = inWave.map((task) => `- ${task.id}${task.children.length ? ` (${task.children.map((c) => c.id).join(", ")})` : ""}: ${task.title}${task.files.length ? ` — files: ${task.files.join(", ")}` : ""}`).join("\n") || "(none — take the next wave)";
1663
+ const progress = taskProgress(execution.tasks);
1664
+ const vcDone = execution.items.filter((item) => item.done).length;
1626
1665
  const mode = resolveGraphMode(ctx?.cwd ?? process.cwd());
1627
1666
  const graphLine =
1628
1667
  mode === "config-unavailable"
1629
- ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi_plans/config.json is repaired]`
1668
+ ? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
1630
1669
  : graphBlockForExecutor(mode === "enabled");
1631
- // F-002 (impl review r1): the next-action line is hoisted out of the
1632
- // implItems ternary so plans without implementation items get the same
1633
- // same-source guidance the panel shows.
1634
- const nextActionLine = `\nSuggested next action (displayed in the pi-plans panel): ${deriveNextAction(execution, executionIsWaiting(execution), resolveImplStatuses(execution.implItems ?? [], execution.items, execution.implStatus), remaining, execution.currentI)}`;
1635
- const implementationItems = execution.implItems?.length
1636
- ? `\nImplementation items: ${execution.implItems.map((item) => item.id).join(", ")}${execution.currentI ? `\nCurrent implementation item: \`${execution.currentI}\`` : ""}\nWhen beginning an implementation item, emit its current anchor exactly once as \`[I-###:current]\`; then use \`[I-###:implemented]\` or \`[I-###:validating]\` for progress.`
1670
+ const rollbackNote = execution.audit.failed.length > 0
1671
+ ? `\nCompletion audit round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
1637
1672
  : "";
1638
1673
  return `[PI-PLANS EXECUTION — write access enabled]
1639
- Implement the accepted plan at ${execution.planPath} (${execution.items.length - remaining.length}/${execution.items.length} verifier items done).
1674
+ Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
1640
1675
 
1641
- Remaining verifier items:
1642
- ${list}${implementationItems}${nextActionLine}
1676
+ Current wave ${currentWave} open tasks:
1677
+ ${waveList}
1678
+
1679
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
1643
1680
 
1644
1681
  ${graphLine}
1645
1682
 
1646
1683
  Execution rules:
1647
- - Implement implementation items in dependency order; grow the change in layers — smallest end-to-end slice first, then stack each new capability on top of what already works.
1648
- - Report implementation-item progress with lightweight markers in your reply: write \`[I-001:implemented]\` when an item's code is done, \`[I-001:validating]\` when you start verifying it. The execution status bar tracks these states.
1649
- - For subprocess-backed verification, when a step starts a subprocess and needs its result before verifying, use literal \`waiting for\` with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s; restart at 5s for each new subprocess.
1650
- - Simplest implementation that fully meets the item: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1651
- - Architectural decisions are for the long term: no stopgaps. Do not add backward-compatibility layers, fallbacks, or migrations — remove the obsolete paths this change obsoletes.
1652
- - Prefer established, well-maintained libraries when they reduce complexity or improve reliability; before writing your own implementation or adding a package, check the project's existing dependencies (docs and types) — never reimplement common functionality without a clear reason.
1653
- - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists; when unsure, skip and emit \`[test skipped: <name>, add when <trigger>]\`.
1654
- - After verifying an item's pass condition with its stated evidence, include \`[DONE:VC-xxx]\` in your reply.
1655
- - When every item is done, report a completion summary.`;
1684
+ - Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
1685
+ - Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
1686
+ - Close subtasks before their parent; a parent is auditable only when every child is terminal.
1687
+ - When every task is terminal, the independent completion auditor verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
1688
+ - Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
1689
+ - Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
1690
+ - Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
1691
+ - MINIMUM tests: trivial one-liners get no test; non-trivial logic gets exactly one minimal check; reuse the repo's test runner when one exists.
1692
+ - For subprocess-backed verification, when a step needs a subprocess result before proceeding, poll with backoff \`5s -> 10s -> 20s -> 40s -> 80s\`, then keep polling at 80s.`;
1656
1693
  }
1657
1694
 
1658
1695
  interface SessionEntry {
@@ -1664,24 +1701,21 @@ interface SessionEntry {
1664
1701
 
1665
1702
  /**
1666
1703
  * Rebuild execution state from the session on start/resume. Finds the last
1667
- * pi-plans-exec snapshot, then re-scans assistant messages after it for
1668
- * [DONE:VC-xxx] markers so progress survives restarts.
1704
+ * pi-plans-exec snapshot; the task tree is rebuilt from the persisted
1705
+ * snapshot (tool-driven progress survives restarts without text replay).
1669
1706
  */
1670
- export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
1671
- pendingExecutionFlush = false; // no flush debt survives a restart
1672
- goalWaitRuntime = null;
1707
+ export async function restoreFromSession(ctx: ExtensionContext, entries: SessionEntry[]): Promise<void> {
1708
+ pendingExecutionFlush = false;
1709
+ continuationRuntime = null;
1673
1710
  resetExecutionCompactionState(ctx);
1674
- let snapshotIndex = -1;
1675
1711
  let snapshot: ExecState | null = null;
1676
1712
  for (let i = entries.length - 1; i >= 0; i--) {
1677
1713
  const entry = entries[i];
1678
1714
  if (entry.type === "custom" && entry.customType === "pi-plans-exec" && entry.data) {
1679
1715
  snapshot = entry.data;
1680
- snapshotIndex = i;
1681
1716
  break;
1682
1717
  }
1683
1718
  if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1684
- // Execution was explicitly stopped or completed after the last snapshot.
1685
1719
  execution = null;
1686
1720
  updateStatusWidget(ctx);
1687
1721
  return;
@@ -1692,78 +1726,48 @@ export async function restoreFromSession(pi: ExtensionAPI, ctx: ExtensionContext
1692
1726
  updateStatusWidget(ctx);
1693
1727
  return;
1694
1728
  }
1695
- // Ignore stale plans whose file vanished.
1696
1729
  if (!fs.existsSync(snapshot.planPath)) {
1697
1730
  execution = null;
1698
1731
  updateStatusWidget(ctx);
1699
1732
  return;
1700
1733
  }
1734
+ // Re-derive the task tree from the plan file (fresh parse) merged with
1735
+ // the snapshot's persisted statuses — a stale parse cannot freeze progress.
1736
+ let tasks: TaskView[];
1737
+ let items: CheckItem[] = snapshot.items.map((item) => ({ ...item }));
1738
+ let planTasks = snapshot.planTasks;
1739
+ try {
1740
+ const planText = fs.readFileSync(snapshot.planPath, "utf8");
1741
+ planTasks = parsePlanTasks(planText);
1742
+ const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
1743
+ tasks = buildTaskView(planTasks, snapshotProgress);
1744
+ items = parseChecklist(planText);
1745
+ } catch {
1746
+ tasks = snapshot.tasks;
1747
+ }
1701
1748
  execution = {
1702
1749
  planPath: snapshot.planPath,
1703
- items: snapshot.items.map((item) => ({ ...item })),
1750
+ items,
1751
+ planTasks,
1752
+ tasks,
1753
+ legacyPlan: snapshot.legacyPlan,
1704
1754
  startedAt: snapshot.startedAt,
1705
1755
  usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
1706
- implItems: snapshot.implItems ?? [],
1707
- implStatus: { ...(snapshot.implStatus ?? {}) },
1708
- // D-008: chrome language is re-resolved at restore time from the
1709
- // CURRENT config rather than trusted from the snapshot, so a
1710
- // `plans set-language` change survives restarts.
1711
1756
  uiLanguage: resolveUiLanguage(ctx.cwd),
1712
- currentI: snapshot.currentI ?? inferCurrentI(snapshot.implItems, snapshot.items, snapshot.implStatus),
1713
- goalWait: snapshot.goalWait
1714
- ? { ...snapshot.goalWait }
1715
- : { noProgressRounds: 0, waitRounds: 0, lastMarkers: null, paused: false },
1757
+ stall: { ...snapshot.stall, lastSnapshot: null, rounds: 0 },
1758
+ audit: { rounds: snapshot.audit?.rounds ?? 0, failed: snapshot.audit?.failed ?? [], running: false },
1759
+ auditLatch: { auditedThisSettle: false, activity: 0 },
1716
1760
  };
1717
- // Distrust the snapshot's implItems: re-parse + re-lint from the plan
1718
- // file so a stale empty list (older parse or format drift at snapshot
1719
- // time) cannot freeze a fake "I 0/0" panel after a restart.
1720
- try {
1721
- const planText = fs.readFileSync(snapshot.planPath, "utf8");
1722
- execution.implItems = parseImplItems(planText);
1723
- execution.implWarning = lintImplItems(planText);
1724
- } catch {
1725
- /* plan file unreadable mid-restore: keep the snapshot values */
1726
- }
1727
- for (let i = snapshotIndex + 1; i < entries.length; i++) {
1728
- const entry = entries[i];
1729
- if (entry.type === "custom" && entry.customType === "pi-plans-exec-cleared") {
1730
- execution = null;
1731
- break;
1732
- }
1733
- const message = entry.message;
1734
- if (message && message.role === "assistant") {
1735
- const text = message.content
1736
- .filter((part) => part.type === "text")
1737
- .map((part) => part.text ?? "")
1738
- .join("\n");
1739
- applyDoneMarkers(text);
1740
- applyImplMarkers(text);
1741
- applyCurrentIMarker(text);
1742
- }
1743
- }
1744
- if (execution) {
1745
- resetGoalWaitRuntime(ctx);
1746
- // Rebind the run identity after a restart so the panel's activity row
1747
- // (and any run-status mirroring) resolves to the active run instead of
1748
- // staying null until the next startExecution (CQ1/D-005 wiring gap).
1749
- const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1750
- executionRunId = active?.run_id ?? null;
1751
- if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1752
- // D-010: replay may have advanced progress past the persisted baseline.
1753
- // Recompute the goal-wait markers; new progress resets the guard counters.
1754
- if (execution.goalWait) {
1755
- const markerSnapshot = goalWaitSnapshot();
1756
- if (markerSnapshot !== execution.goalWait.lastMarkers) {
1757
- execution.goalWait.lastMarkers = markerSnapshot;
1758
- execution.goalWait.noProgressRounds = 0;
1759
- execution.goalWait.waitRounds = 0;
1760
- }
1761
- }
1762
- persist(pi); // refresh snapshot so the next resume has less to rescan
1763
- if (isExecutionComplete()) {
1764
- // Completed during the rescan: restore the planning model on the way out.
1765
- await completeExecution(pi, ctx);
1766
- }
1761
+ execution.stall.lastSnapshot = stallSnapshot();
1762
+ resetContinuationRuntime(ctx);
1763
+ const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
1764
+ executionRunId = active?.run_id ?? null;
1765
+ if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
1766
+ persist(ctx);
1767
+ if (pendingAudit()) {
1768
+ // Terminal tasks without a passing audit: rerun the audit flow.
1769
+ latchAuditThisSettle();
1770
+ await runAuditFlow(ctx);
1767
1771
  }
1768
1772
  updateStatusWidget(ctx);
1769
1773
  }