pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
@@ -1,9 +1,12 @@
1
1
  /**
2
2
  * `analyze_refs` tool — plan-with-refs per-reference analysis via read-only Pi
3
3
  * subagents with isolated context. One lane per reference (cwd = the ref's own
4
- * directory), reusing the reviewer role gates from `.git/pi_plans/config.json`
5
- * and the concurrent refinement overlay (title "Refs"). Batches are capped at
6
- * three concurrent lanes; larger ref sets run as sequential batches.
4
+ * directory), reusing the reviewer model confirmation from the GLOBAL config
5
+ * (`~/.pi/pi-plans/config.json`) and the concurrent overlay (title "Refs").
6
+ * analyze_refs is spawn-only by nature, so the reviewer MODE is deliberately
7
+ * not consulted here (Q-4=B): a current-session reviewer still gets spawned
8
+ * ref-analyst lanes, with a one-time notice in the result. Batches are capped
9
+ * at three concurrent lanes; larger ref sets run as sequential batches.
7
10
  *
8
11
  * Recording is best-effort: spawns land in `subagents.jsonl` (role
9
12
  * `ref-analyst`) only when an active planning run exists. Analysis output is
@@ -17,7 +20,18 @@ import { Text } from "@earendil-works/pi-tui";
17
20
  import { Type } from "typebox";
18
21
  import * as fs from "node:fs";
19
22
  import * as path from "node:path";
20
- import { loadConfig, normalizeWorkdir, readActive, recordSubagent, resolveStateRootOrNull, StateError } from "../src/state.ts";
23
+ import {
24
+ loadConfig,
25
+ normalizeWorkdir,
26
+ readActive,
27
+ recordSubagent,
28
+ resolveEffectiveReviewer,
29
+ resolveGlobalConfigPath,
30
+ resolveStateRootOrNull,
31
+ StateError,
32
+ } from "../src/state.ts";
33
+ import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type RolePanelHost } from "../src/role-panels.ts";
34
+ import { roleModelLabel } from "../src/thinking-levels.ts";
21
35
  import type { SubagentUsage } from "../src/subagent.ts";
22
36
  import { resolveActiveRun } from "../src/run-context.ts";
23
37
  import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
@@ -45,25 +59,39 @@ const AnalyzeRefsParams = Type.Object({
45
59
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
46
60
  });
47
61
 
48
- function gateError(problem: "state" | "mode" | "current-session" | "confirm"): StateError {
62
+ function gateError(problem: "state" | "confirm", guidance?: string): StateError {
49
63
  if (problem === "state") {
50
64
  return new StateError("no pi-plans state found; run the plans tool (action: init) first");
51
65
  }
52
- if (problem === "mode") {
53
- return new StateError(
54
- "The reviewer role mode is missing or invalid in .git/pi_plans/config.json (analyze_refs reuses the reviewer gates). Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer).",
55
- );
56
- }
57
- if (problem === "current-session") {
58
- return new StateError(
59
- "The reviewer role mode is current-session, but analyze_refs only spawns delegated read-only subagents (one per reference). Ask the user to switch the reviewer mode to delegated-subagent via ask_choice, persist with the plans tool (set-role, role=reviewer, mode=delegated-subagent), then retry analyze_refs.",
60
- );
61
- }
62
66
  return new StateError(
63
- "The reviewer model was never confirmed (confirmed_at is null); analyze_refs reuses the reviewer confirmation. Ask the model-confirmation question with ask_choice: 1. Inherit the main agent's model (recommended) 2. Choose a model (list options from the /model picker; persist the exact provider/model selector) 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer, confirmed: true, modelSelector: the selector or 'inherit').",
67
+ guidance ??
68
+ firstUseTextGuidance([], resolveGlobalConfigPath()),
64
69
  );
65
70
  }
66
71
 
72
+ /** First-use model confirmation for the spawn-only ref-analyst path: native
73
+ * panels in TUI, menus for hasUI non-TUI, embedded text guidance otherwise.
74
+ * The reviewer MODE is not consulted (Q-4=B), but because analysis always
75
+ * spawns, a confirmed CONCRETE model is required even when the stored mode
76
+ * is current-session (model confirmation “as usual”). */
77
+ async function ensureRefAnalystModelReady(
78
+ host: RolePanelHost,
79
+ role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
80
+ ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
81
+ if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
82
+ let outcome = await runFirstUseFlow(host, role.thinking_level);
83
+ if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
84
+ // F-008: a manually entered selector that the registry does not know —
85
+ // one re-pick, then let spawn-side errors surface precisely.
86
+ outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
87
+ }
88
+ if (outcome.status === "cancelled") throw firstUseCancelledError("analyze_refs");
89
+ const guidance = firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath());
90
+ if (outcome.status === "unavailable") throw gateError("confirm", guidance);
91
+ if (outcome.role.model_selector === null) throw gateError("confirm", guidance);
92
+ return outcome.role;
93
+ }
94
+
67
95
  interface AnalysisJob {
68
96
  input: RefAnalystTaskInput;
69
97
  name: string;
@@ -72,14 +100,14 @@ interface AnalysisJob {
72
100
  missing: string | null;
73
101
  }
74
102
 
75
- export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void {
103
+ export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): void {
76
104
  const agentPrompt = stripFrontmatter(fs.readFileSync(path.join(baseDir, "agents", "ref-analyst.md"), "utf8"));
77
105
 
78
- pi.registerTool({
106
+ ext.registerTool({
79
107
  name: "analyze_refs",
80
108
  label: "Analyze Refs",
81
109
  description:
82
- "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer role gates and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. Refuses until the reviewer mode/model gates pass in .git/pi_plans/config.json.",
110
+ "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. The reviewer mode is not consulted (spawn-only); first use pops native model/effort panels in TUI.",
83
111
  promptSnippet: "Analyze plan-with-refs references with per-ref read-only subagents",
84
112
  parameters: AnalyzeRefsParams,
85
113
 
@@ -92,17 +120,22 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
92
120
  throw gateError("state");
93
121
  }
94
122
  const config = loadConfig(root);
95
- const reviewer = config.reviewer;
96
- if (!reviewer || (reviewer.mode !== "delegated-subagent" && reviewer.mode !== "current-session")) {
97
- throw gateError("mode");
98
- }
99
- if (reviewer.mode === "current-session") {
100
- throw gateError("current-session");
101
- }
102
- if (reviewer.confirmed_at === null) {
103
- throw gateError("confirm");
123
+
124
+ // F-005: cheap validations BEFORE any first-use panel.
125
+ if (params.refs.length === 0) {
126
+ throw new StateError("analyze_refs requires at least one reference");
104
127
  }
105
128
 
129
+ // Effective reviewer from the global config (mode NOT consulted —
130
+ // analyze_refs is spawn-only, Q-4=B; a notice surfaces when the stored
131
+ // mode is current-session so the switch is never silent).
132
+ const { reviewer: initialReviewer } = resolveEffectiveReviewer(root);
133
+ const modeIgnoredNotice =
134
+ initialReviewer.mode === "current-session"
135
+ ? `note: the reviewer mode is ${initialReviewer.mode}, but analyze_refs always spawns read-only subagents; the mode is ignored here and unchanged.`
136
+ : null;
137
+ const reviewer = await ensureRefAnalystModelReady(ctx as unknown as RolePanelHost, initialReviewer);
138
+
106
139
  // Resolve refs and validate directories up front; missing ones become
107
140
  // FAILED sections instead of aborting the whole batch.
108
141
  const active = resolveActiveRun(ctx.sessionManager, workdir);
@@ -122,7 +155,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
122
155
  }
123
156
 
124
157
  const model = reviewer.model_selector ?? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined);
125
- const modelLabel = model ?? "inherit";
158
+ const modelLabel = roleModelLabel(model ?? "inherit", reviewer.thinking_level);
126
159
  let languageTag: string | null = null;
127
160
  if (active) {
128
161
  try {
@@ -140,6 +173,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
140
173
  role: "ref-analyst",
141
174
  name,
142
175
  model: okModel ?? model ?? null,
176
+ thinking_level: reviewer.thinking_level,
143
177
  // I-010: meter subagent token/cost for benchmark accounting.
144
178
  usage: usage
145
179
  ? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
@@ -160,6 +194,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
160
194
  task: buildRefAnalystTask({ ...job.input, languageTag }),
161
195
  cwd: job.dir,
162
196
  model,
197
+ thinkingLevel: reviewer.thinking_level ?? undefined,
163
198
  tools: READ_ONLY_TOOLS,
164
199
  signal: relay.signal,
165
200
  onProgress: (event) => overlay?.update(job.laneId, event),
@@ -224,7 +259,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
224
259
 
225
260
  if (failures === jobs.length) {
226
261
  throw new Error(
227
- `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) and re-ask the model-confirmation question.` : ""}`,
262
+ `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next analyze_refs opens the native model panel to re-confirm.` : ""}`,
228
263
  );
229
264
  }
230
265
 
@@ -237,13 +272,13 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
237
272
  content: [
238
273
  {
239
274
  type: "text",
240
- text: `${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
275
+ text: `${modeIgnoredNotice ? `${modeIgnoredNotice}\n\n` : ""}${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
241
276
  },
242
277
  ],
243
278
  details: {
244
279
  mode: "delegated-subagent",
245
280
  role: "ref-analyst",
246
- reviewerGates: { mode: reviewer.mode, model },
281
+ reviewerGates: { mode: initialReviewer.mode, modeIgnored: modeIgnoredNotice !== null, model, thinkingLevel: reviewer.thinking_level },
247
282
  batches: Math.ceil(jobs.length / BATCH_SIZE),
248
283
  model,
249
284
  outputs,
@@ -16,13 +16,6 @@ import { Text } from "@earendil-works/pi-tui";
16
16
  import { Type } from "typebox";
17
17
  import { disableAutoComplete, enableAutoComplete, isAutoCompleteEnabled, recordAskChoice } from "../src/autocomplete.ts";
18
18
  import { assertAutoApprovable, isAutoApproveEnabled } from "../src/auto-approve.ts";
19
- import {
20
- TERMINATION_QUESTION,
21
- TERMINATION_OPTIONS,
22
- TERMINATION_RECORDING_INSTRUCTIONS,
23
- implReviewerCountPromptLine,
24
- renderTerminationOptions,
25
- } from "../src/termination-prompt.ts";
26
19
  import { truncateToWidth, visibleWidth } from "../src/refine-ui-helpers.ts";
27
20
  import { stripRecommendedMarker } from "../src/ask-form.ts";
28
21
  import {
@@ -63,8 +56,8 @@ export const FALLBACK_ROWS = 30;
63
56
  /** Minimal-form floor for tiny terminals (stage-3 width). */
64
57
  const MINIMAL_LINE_WIDTH = 20;
65
58
  /**
66
- * Truncation floor for fixed tail labels (Other…/Auto-complete/Auto-refine
67
- * loop): the longest magic prefix ("Auto-refine loop", 16 cols) plus slack.
59
+ * Truncation floor for fixed tail labels (Other…/Auto-complete): the
60
+ * longest magic prefix ("Auto-complete", 14 cols) plus slack.
68
61
  * These labels drive startsWith() answer routing and must never lose it.
69
62
  */
70
63
  const FIXED_LABEL_FLOOR = 18;
@@ -74,7 +67,7 @@ export interface PanelItem {
74
67
  core: string;
75
68
  /** Full display label: core + description (degradation stage 0). */
76
69
  display: string;
77
- /** Fixed tail labels (Other…/Auto-complete/Auto-refine loop): truncation keeps at least the magic prefix. */
70
+ /** Fixed tail labels (Other…/Auto-complete): truncation keeps at least the magic prefix. */
78
71
  fixed?: boolean;
79
72
  }
80
73
 
@@ -213,12 +206,6 @@ export const AskChoiceParams = Type.Object(
213
206
  purpose: Type.Optional(
214
207
  Type.String({ description: "Short machine-readable purpose (e.g. 'scope', 'termination-condition')." }),
215
208
  ),
216
- trailing: Type.Optional(
217
- StringEnum(["auto-refine-loop"] as const, {
218
- description:
219
- 'Replace the trailing Auto-complete option with "Auto-refine loop" (post-execution amelioration prompt). Selecting it returns instructions to ask the rounds/termination follow-up; Auto-complete is suppressed entirely for this question.',
220
- }),
221
- ),
222
209
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
223
210
  },
224
211
  { additionalProperties: false },
@@ -589,12 +576,12 @@ function formatBatchAnswers(batch: NonNullable<AskChoiceDetails["batch"]>): stri
589
576
  }
590
577
  const NL = "\n";
591
578
 
592
- export function registerAskChoiceTool(pi: ExtensionAPI): void {
593
- pi.registerTool({
579
+ export function registerAskChoiceTool(ext: ExtensionAPI): void {
580
+ ext.registerTool({
594
581
  name: "ask_choice",
595
582
  label: "Ask Choice",
596
583
  description:
597
- "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt. EVERY option you author — including the accept/execute handoff and the implementation-review setup questions — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
584
+ "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the questionIds reserved for handoff. EVERY option you author — including the accept/execute handoff — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
598
585
  promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
599
586
  promptGuidelines: [
600
587
  "Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
@@ -605,13 +592,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
605
592
  executionMode: "sequential",
606
593
 
607
594
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
608
- // R-13 (defense-in-depth): a delegated executor child has no user to
609
- // answer — refuse instead of blocking a headless run on a UI prompt.
610
- if (process.env.PI_PLANS_EXECUTOR === "1") {
611
- throw new Error(
612
- "ask_choice is unavailable in a delegated executor session: no interactive user. Decide autonomously, proceed, and record the deviation in your final summary.",
613
- );
614
- }
615
595
  // 0.4.0 batch mode: one tabbed form for a whole round of questions
616
596
  // (2-8). The single-question path below is untouched (C-004).
617
597
  // F-005 (impl review r1): ambiguous shapes fail loudly instead of
@@ -620,9 +600,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
620
600
  if (params.question !== undefined || params.options !== undefined) {
621
601
  throw new Error("ask_choice accepts either question+options or questions, not both");
622
602
  }
623
- if (params.trailing !== undefined) {
624
- throw new Error("ask_choice batch mode does not support trailing (single-question only)");
625
- }
626
603
  return executeAskChoiceBatch({ questions: params.questions, workdir: params.workdir }, ctx);
627
604
  }
628
605
  const workdir = normalizeWorkdir(params.workdir ?? ctx.cwd);
@@ -661,10 +638,7 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
661
638
  /* the decisions ledger already holds the answer; F-005 reconcile covers the gap */
662
639
  }
663
640
  };
664
- // Param normalization: a trailing option replaces Auto-complete entirely,
665
- // so an erroneously passed autoComplete flag is suppressed here.
666
- const trailing = params.trailing;
667
- const autoComplete = (params.autoComplete ?? true) && trailing === undefined;
641
+ const autoComplete = params.autoComplete ?? true;
668
642
  const options = params.options;
669
643
  if (options.length === 0) throw new Error("ask_choice requires at least one option");
670
644
  const recommended = options.find((option) => option.recommended) ?? options[0];
@@ -764,8 +738,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
764
738
  };
765
739
  }
766
740
 
767
- const AUTO_REFINE_LOOP_LABEL =
768
- "Auto-refine loop (run refinement rounds until no high-severity finding or the 5-round cap)";
769
741
  const panelItems: PanelItem[] = options.map((option, index) => {
770
742
  const label = stripRecommendedMarker(option.label);
771
743
  const isRec = option === recommended;
@@ -777,7 +749,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
777
749
  });
778
750
  if (allowOther) panelItems.push({ core: "Other… (type your own answer)", display: "Other… (type your own answer)", fixed: true });
779
751
  if (autoComplete) panelItems.push({ core: "Auto-complete (take the recommended option)", display: "Auto-complete (take the recommended option)", fixed: true });
780
- else if (trailing) panelItems.push({ core: AUTO_REFINE_LOOP_LABEL, display: AUTO_REFINE_LOOP_LABEL, fixed: true });
781
752
 
782
753
  const panel = fitAskChoicePanel(
783
754
  params.question,
@@ -820,23 +791,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
820
791
  };
821
792
  }
822
793
 
823
- if (trailing && selected.startsWith("Auto-refine loop")) {
824
- recordAskChoice(ctx, false);
825
- record("Auto-refine loop", "user");
826
- // Skill-aware reviewer-count default (D-1/D-4): same mapping the
827
- // goal-running continuation in src/exec.ts renders.
828
- const activeSkill = resolveActiveRun(ctx.sessionManager, workdir)?.skill;
829
- return {
830
- content: [
831
- {
832
- type: "text",
833
- text: `User selected Auto-refine loop. Immediately ask the follow-up with ask_choice (autoComplete: false, in the session language): "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(activeSkill)} Then run the loop per the completion instructions: each round calls refine (role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>), accepts findings on evidence, applies fixes, re-runs relevant tests, and continues until the chosen termination condition — the goal-wait option keeps the loop running until no unpassed VCs remain.`,
834
- },
835
- ],
836
- details: details("Auto-refine loop", "user"),
837
- };
838
- }
839
-
840
794
  if (allowOther && selected.startsWith("Other…")) {
841
795
  const typed = await ctx.ui.input(`${params.question} — your answer:`);
842
796
  if (typed === undefined || !typed.trim()) {
@@ -128,8 +128,8 @@ export async function ensureRuntime(workdir: string, ctx: CodeGraphContext): Pro
128
128
  return { entry: runtimeCache, status };
129
129
  }
130
130
 
131
- export function registerCodeGraphTool(pi: ExtensionAPI): void {
132
- pi.registerTool({
131
+ export function registerCodeGraphTool(ext: ExtensionAPI): void {
132
+ ext.registerTool({
133
133
  name: "code_graph",
134
134
  label: "Code Graph",
135
135
  description:
@@ -1,6 +1,9 @@
1
1
  /**
2
- * `execute_plan` tool — the execution handoff. On explicit user approval (no
3
- * Auto-complete) the extension enters execution mode with checklist tracking.
2
+ * `execute_plan` tool — the execution handoff (v0.6.1). On explicit user
3
+ * approval (no Auto-complete) the extension enters task-tree execution mode:
4
+ * progress flows through `plans_update_task`, the dashboard tracks every
5
+ * task, and the execution reviewer gates the final pass. Legacy I-### plans
6
+ * parse through the fallback with an upgrade notice.
4
7
  */
5
8
 
6
9
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
@@ -12,15 +15,13 @@ import {
12
15
  getExecution,
13
16
  resumeActiveExecution,
14
17
  startExecution,
15
- type ExecutionRuntime,
16
18
  } from "../src/exec.ts";
17
19
  import { disableAutoComplete } from "../src/autocomplete.ts";
18
20
  import { isAutoApproveEnabled } from "../src/auto-approve.ts";
19
- import { latestPlanVersion, parseChecklist, parseImplItems } from "../src/plan.ts";
20
- import { normalizeWorkdir, recordDecision, type RunSummary } from "../src/state.ts";
21
- import { bindRun, resolveActiveRun } from "../src/run-context.ts";
21
+ import { checklistHeaderName, latestPlanVersion, lintPlanTasks, parseChecklist, parsePlanTasks } from "../src/plan.ts";
22
+ import { normalizeWorkdir, type RunSummary } from "../src/state.ts";
23
+ import { bindRun } from "../src/run-context.ts";
22
24
  import { executionCandidates, resolveCommandRun } from "../src/run-picker.ts";
23
- import { collectModelSelectors, modelSelectorOf } from "../src/config-command.ts";
24
25
  import { resolveUiLanguage } from "../src/ui-language.ts";
25
26
 
26
27
 
@@ -43,7 +44,7 @@ export async function executeHandoff(
43
44
  ctx: ExtensionContext,
44
45
  planPathArg?: string,
45
46
  workdirArg?: string,
46
- signal?: AbortSignal,
47
+ _signal?: AbortSignal,
47
48
  ): Promise<HandoffOutcome> {
48
49
  const workdir = normalizeWorkdir(workdirArg ?? ctx.cwd);
49
50
 
@@ -79,10 +80,28 @@ export async function executeHandoff(
79
80
  if (items.length === 0) {
80
81
  return {
81
82
  status: "error",
82
- message: `${planPath} has no parsable \`## Verifier Checklist\` with \`- [ ] \`VC-###\` ...\` items. Fix the plan before execution.`,
83
+ message: `${planPath} has no parsable \`## Verification Checks\` (or legacy \`## Verifier Checklist\`) with \`- [ ] \`VC-###\` ...\` items. Fix the plan before execution.`,
83
84
  };
84
85
  }
85
- const implItems = parseImplItems(planText);
86
+ const planTasks = parsePlanTasks(planText);
87
+ if (planTasks.tasks.length === 0) {
88
+ return {
89
+ status: "error",
90
+ message: `${planPath} has no parsable tasks: add a \`## Tasks\` section (\`- \`Task-1\`: title — deps: …; files: …; wave: 1\`).`,
91
+ };
92
+ }
93
+ // I-001/R-001: task-tree consistency is advisory while planning and
94
+ // hard-rejected at this gate. Legacy I-### fallback plans are exempt
95
+ // (their shape predates the microsyntax).
96
+ if (!planTasks.legacy) {
97
+ const lint = lintPlanTasks(planText);
98
+ if (lint !== null) {
99
+ return {
100
+ status: "error",
101
+ message: `${planPath} failed the task-tree consistency gate; fix these before execution:\n${lint}`,
102
+ };
103
+ }
104
+ }
86
105
 
87
106
  disableAutoComplete(ctx, "execution handoff");
88
107
  // I-004/D-019: PI_PLANS_AUTO_APPROVE=1 short-circuits the confirm BEFORE
@@ -96,14 +115,22 @@ export async function executeHandoff(
96
115
  };
97
116
  }
98
117
 
118
+ const legacyPlan = planTasks.legacy || checklistHeaderName(planText) === "Verifier Checklist";
119
+
99
120
  let approved: boolean;
100
121
  if (autoApprove) {
101
122
  approved = true;
102
123
  } else {
124
+ const lang = resolveUiLanguage(workdir);
103
125
  const preview = items.map((item) => `- ${item.done ? "☑" : "☐"} ${item.id}`).join("\n");
126
+ const legacyNote = legacyPlan
127
+ ? (lang === "zh"
128
+ ? "\n\n注意:该计划使用旧版 I-### 格式,将以兼容映射执行;建议在下次修订时升级为 ## Tasks 新格式。"
129
+ : "\n\nNote: this plan uses the legacy I-### format and executes through the compatibility mapping; upgrade it to the ## Tasks format at the next revision.")
130
+ : "";
104
131
  approved = await ctx.ui.confirm(
105
132
  "Execute this plan now?",
106
- `${planPath}\n${items.length} verifier item(s):\n${preview}\n\nExecution mode enables write access and tracks [DONE:VC-xxx] progress.`,
133
+ `${planPath}\n${items.length} verification check(s) over ${planTasks.tasks.length} task(s):\n${preview}${legacyNote}\n\nExecution mode enables write access; task progress is reported with the plans_update_task tool and gated by the execution reviewer.`,
107
134
  );
108
135
  }
109
136
  if (!approved) {
@@ -114,81 +141,15 @@ export async function executeHandoff(
114
141
  // the approval checkpoint and status flip land on the run the user chose.
115
142
  if (chosenRun) bindRun(ctx.sessionManager, workdir, chosenRun.run_id);
116
143
 
117
- // v0.6.0 (R-8): runtime question — current session (recommended) or a
118
- // delegated executor on another model. Skipped under auto-approve/no-UI.
119
- const runtime = await chooseExecutionRuntime(ctx, workdir, autoApprove, chosenRun);
120
-
121
- await startExecution(getCurrentApi(), ctx, planPath, items, implItems, { runtime, signal });
122
- const scopeNote = implItems.length ? ` Tracking ${implItems.length} implementation item(s).` : "";
144
+ await startExecution(ctx, { planPath, planTasks, items });
123
145
  const autoNote = autoApprove ? "[auto-approve] " : "";
124
- const runtimeNote = runtime === "current-session" ? "" : ` Delegated to executor model ${runtime.modelSelector}; progress mirrors in the overlay.`;
146
+ const legacyNote = legacyPlan ? " Legacy I-### mapping active; upgrade the plan at the next revision." : "";
125
147
  return {
126
148
  status: "executing",
127
149
  planPath,
128
150
  itemCount: items.length,
129
- message: `${autoNote}Execution approved. ${items.length} verifier item(s) queued; implement in dependency order and mark verified items with [DONE:VC-xxx].${scopeNote}${runtimeNote}`,
130
- };
131
- }
132
-
133
- const MODEL_SELECTOR_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*\/[A-Za-z0-9][A-Za-z0-9._-]*$/;
134
-
135
- /**
136
- * R-8: ask where the execution runs. Uses ctx.ui.select directly (ask_choice
137
- * is a tool and cannot be invoked from tool/command context). Under
138
- * auto-approve or no-UI the question is skipped: current session, decision
139
- * recorded with the [auto-approve] annotation convention.
140
- */
141
- async function chooseExecutionRuntime(
142
- ctx: ExtensionContext,
143
- workdir: string,
144
- autoApprove: boolean,
145
- chosenRun: RunSummary | null,
146
- ): Promise<ExecutionRuntime> {
147
- const record = (answer: string, source: "user" | "auto-complete", question: string, options: string[]): void => {
148
- const runId = chosenRun?.run_id ?? resolveActiveRun(ctx.sessionManager, workdir)?.run_id ?? null;
149
- if (!runId) return;
150
- try {
151
- recordDecision(workdir, runId, {
152
- question,
153
- options,
154
- answer,
155
- answer_source: source,
156
- });
157
- } catch {
158
- /* decision audit trail is best-effort */
159
- }
151
+ message: `${autoNote}Execution approved. ${planTasks.tasks.length} task(s) queued in wave order; report progress with the plans_update_task tool (status + evidence); the execution reviewer verifies every check before the run completes.${legacyNote}`,
160
152
  };
161
- if (autoApprove || !ctx.hasUI) {
162
- record("current session [auto-approve]", "auto-complete", "Execution runtime", ["current session", "switch model"]);
163
- return "current-session";
164
- }
165
- const lang = resolveUiLanguage(workdir);
166
- const currentLabel = lang === "zh" ? "使用当前会话(推荐)" : "Use the current session (recommended)";
167
- const switchLabel = lang === "zh" ? "切换至其他模型…" : "Switch to another model…";
168
- const title = lang === "zh" ? "执行运行时" : "Execution runtime";
169
- const first = await ctx.ui.select(title, [currentLabel, switchLabel]);
170
- if (first === undefined || first === currentLabel) {
171
- record("current session", "user", "Execution runtime", [currentLabel, switchLabel]);
172
- return "current-session";
173
- }
174
- // Model picker: switch targets exclude the current selector by design
175
- // (option 1 IS the current session).
176
- const currentSelector = modelSelectorOf(ctx.model);
177
- const targets = collectModelSelectors(ctx, currentSelector);
178
- const otherLabel = lang === "zh" ? "其他(输入 provider/model)…" : "Other (type provider/model)…";
179
- const modelTitle = lang === "zh" ? "切换至哪个模型执行?" : "Switch to which model?";
180
- let modelPick = await ctx.ui.select(modelTitle, [...targets, otherLabel]);
181
- if (modelPick === otherLabel) {
182
- const typed = await ctx.ui.input(modelTitle, "provider/model");
183
- modelPick = typed && MODEL_SELECTOR_RE.test(typed.trim()) ? typed.trim() : undefined;
184
- }
185
- if (modelPick === undefined || !MODEL_SELECTOR_RE.test(modelPick)) {
186
- // Cancelled or invalid: fall back to the current session, recorded.
187
- record("current session (model switch cancelled)", "user", "Execution runtime", [currentLabel, switchLabel]);
188
- return "current-session";
189
- }
190
- record(`switch model: ${modelPick}`, "user", "Execution runtime", [currentLabel, switchLabel, ...targets, otherLabel]);
191
- return { modelSelector: modelPick };
192
153
  }
193
154
 
194
155
  /** The user command may resume an approved execution; the tool always asks. */
@@ -196,40 +157,35 @@ export async function executeCommand(ctx: ExtensionContext, planPathArg?: string
196
157
  const activeExecution = getExecution();
197
158
  const planPath = planPathArg ? path.resolve(ctx.cwd, planPathArg.replace(/^@/, "")) : activeExecution?.planPath;
198
159
  if (activeExecution && planPath && path.resolve(activeExecution.planPath) === path.resolve(planPath)) {
199
- const resumed = resumeActiveExecution(getCurrentApi(), ctx);
160
+ const resumed = resumeActiveExecution(ctx);
161
+ // v0.8 phase-aware response: a verifying run continues its review loop
162
+ // (this tool is also the ONLY budget-granting surface at a cap pause).
163
+ const statusText = activeExecution.review?.inFlight
164
+ ? "Execution review in progress (status verifying); the reviewer round runs in the overlay."
165
+ : (getExecution()?.stall.paused ?? false)
166
+ ? "Execution review paused at the round cap — this confirmation granted a fresh five-round budget; the review resumes now."
167
+ : "Execution resumed; task progress preserved.";
200
168
  return {
201
169
  status: "executing",
202
170
  planPath,
203
171
  itemCount: activeExecution.items.length,
204
- message: resumed ? "Execution resumed; verified progress preserved." : "This plan is already executing.",
172
+ message: resumed ? statusText : "This plan is already executing.",
205
173
  };
206
174
  }
207
175
  return executeHandoff(ctx, planPathArg);
208
176
  }
209
177
 
210
- // The tool registers with the ExtensionAPI in scope; keep a module-level
211
- // reference so the shared handoff helper can reach appendEntry/sendMessage.
212
- let currentApi: ExtensionAPI | null = null;
213
- export function setCurrentApi(api: ExtensionAPI): void {
214
- currentApi = api;
215
- }
216
- function getCurrentApi(): ExtensionAPI {
217
- if (!currentApi) throw new Error("execute_plan used before extension initialization");
218
- return currentApi;
219
- }
220
-
221
- export function registerExecutePlanTool(pi: ExtensionAPI): void {
222
- setCurrentApi(pi);
223
- pi.registerTool({
178
+ export function registerExecutePlanTool(ext: ExtensionAPI): void {
179
+ ext.registerTool({
224
180
  name: "execute_plan",
225
181
  label: "Execute Plan",
226
182
  description:
227
- "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then asks which runtime executes the plan — the current session (recommended) or a delegated executor subagent on another model (>=3 switch targets listed; the child writes natively and reports [DONE:VC-xxx] markers the parent tracks). Either way the extension tracks Verifier-Checklist progress. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
183
+ "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters task-tree execution mode: every task's progress is reported via the plans_update_task tool (status + evidence), the task dashboard tracks the tree (Ctrl+Shift+T expands it), and an independent execution reviewer verifies the verification checks before the run completes. Legacy I-### plans parse through the compatibility mapping with an upgrade notice. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
228
184
  promptSnippet: "Hand an accepted plan off to the tracked execution loop",
229
185
  parameters: ExecutePlanParams,
230
186
 
231
- async execute(_toolCallId, params, signal, _onUpdate, ctx) {
232
- const outcome = await executeHandoff(ctx, params.planPath, params.workdir, signal);
187
+ async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
188
+ const outcome = await executeHandoff(ctx, params.planPath, params.workdir);
233
189
  if (outcome.status === "error") throw new Error(outcome.message);
234
190
  return {
235
191
  content: [{ type: "text", text: outcome.message }],
@@ -277,10 +277,6 @@ function createGraphReadTool(cwd: string) {
277
277
  };
278
278
  };
279
279
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
280
- // Delegated executor children (PI_PLANS_EXECUTOR=1) always use the
281
- // native tools: DB-first staging would never be materialized inside
282
- // the child (no code_graph in its allowlist), so writes must hit disk.
283
- if (process.env.PI_PLANS_EXECUTOR === "1") return native(null);
284
280
  if (mode === "off") return native(null);
285
281
  if (mode === "config-unavailable") return native("config read failed");
286
282
  const ensured = await ensureRuntime(ctx.cwd, ctx);
@@ -331,7 +327,6 @@ function createGraphWriteTool(cwd: string) {
331
327
  };
332
328
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
333
329
  // Delegated executor children bypass DB-first staging (see read tool).
334
- if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
335
330
  if (mode === "off") return stage(null);
336
331
  if (mode === "config-unavailable") return stage("config read failed");
337
332
  const ensured = await ensureRuntime(ctx.cwd, ctx);
@@ -382,7 +377,6 @@ function createGraphEditTool(cwd: string) {
382
377
  };
383
378
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
384
379
  // Delegated executor children bypass DB-first staging (see read tool).
385
- if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
386
380
  if (mode === "off") return stage(null);
387
381
  if (mode === "config-unavailable") return stage("config read failed");
388
382
  const ensured = await ensureRuntime(ctx.cwd, ctx);
@@ -453,9 +447,9 @@ export function createGraphAwareFileTools(cwd: string) {
453
447
  return tools;
454
448
  }
455
449
 
456
- export function registerGraphAwareFileTools(pi: ExtensionAPI, cwd = process.cwd()): void {
450
+ export function registerGraphAwareFileTools(ext: ExtensionAPI, cwd = process.cwd()): void {
457
451
  const tools = createGraphAwareFileTools(cwd);
458
- pi.registerTool(tools.read);
459
- pi.registerTool(tools.write);
460
- pi.registerTool(tools.edit);
452
+ ext.registerTool(tools.read);
453
+ ext.registerTool(tools.write);
454
+ ext.registerTool(tools.edit);
461
455
  }