pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
package/tools/refine.ts CHANGED
@@ -1,20 +1,35 @@
1
1
  /**
2
- * `refine` tool — reviewer/criticizer refinement rounds via read-only Pi
3
- * subagents with isolated context.
2
+ * `refine` tool — reviewer refinement rounds via read-only Pi subagents
3
+ * with isolated context.
4
4
  *
5
- * Enforces the role-confirmation gate: refuses to spawn while a
6
- * role's mode is invalid or its model was never confirmed, telling the caller
7
- * exactly which ask_choice question to ask first.
5
+ * Enforces the reviewer role gates: the mode question stays agent-mediated
6
+ * (ask_choice), while first-use model confirmation pops native panels in
7
+ * TUI (v0.7.0): a /model-style searchable panel, then a /thinking-style
8
+ * effort panel, persisted to the GLOBAL reviewer config. Esc cancels the
9
+ * whole gate with a dedicated error (details.cancelled) — do not re-ask.
10
+ * The reviewer output carries findings AND questions (Q-###); the caller
11
+ * must ask every question with ask_choice before revising the plan.
8
12
  */
9
13
 
10
- import { StringEnum } from "@earendil-works/pi-ai";
14
+ import { Type } from "typebox";
11
15
  import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
12
16
  import { truncateHead } from "@earendil-works/pi-coding-agent";
13
17
  import { Text } from "@earendil-works/pi-tui";
14
- import { Type } from "typebox";
15
18
  import * as fs from "node:fs";
16
19
  import * as path from "node:path";
17
- import { loadConfig, normalizeWorkdir, readActive, recordSubagent, resolveStateRootOrNull, StateError, type RoleConfig } from "../src/state.ts";
20
+ import {
21
+ loadConfig,
22
+ normalizeWorkdir,
23
+ readActive,
24
+ recordSubagent,
25
+ resolveEffectiveReviewer,
26
+ resolveGlobalConfigPath,
27
+ resolveStateRootOrNull,
28
+ reviewerReady,
29
+ StateError,
30
+ } from "../src/state.ts";
31
+ import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type FirstUseOutcome, type RolePanelHost } from "../src/role-panels.ts";
32
+ import { roleModelLabel } from "../src/thinking-levels.ts";
18
33
  import { uiLanguageFromTag, type UiLanguage } from "../src/ui-language.ts";
19
34
  import type { SubagentUsage } from "../src/subagent.ts";
20
35
  import { resolveActiveRun } from "../src/run-context.ts";
@@ -25,27 +40,20 @@ import {
25
40
  reusableLaneOutputs,
26
41
  startReviewRound,
27
42
  } from "../src/workflow-state.ts";
28
- import { buildCriticizerTask, buildImplementationCriticizerTask, buildImplementationReviewerTask, buildReviewerTask, reviewerLanes } from "../src/refine-prompts.ts";
43
+ import { buildReviewerTask, reviewerLanes } from "../src/refine-prompts.ts";
29
44
  import { graphBlockForRefiner } from "../src/code-graph/prompts.ts";
30
45
  import { runPiSubagent, stripFrontmatter } from "../src/subagent.ts";
31
46
  import { RefineOverlayController, refineOverlayContext } from "../src/refine-ui.ts";
32
47
 
33
48
 
34
49
  const RefineParams = Type.Object({
35
- role: StringEnum(["reviewer", "criticizer"] as const, { description: "Refinement role to run" }),
36
50
  planPath: Type.String({ description: "Path to the PLAN_vN.md to review (absolute or relative to workdir)" }),
37
- target: Type.Optional(
38
- StringEnum(["plan", "implementation"] as const, {
39
- description:
40
- 'Review target: "plan" (default) reviews the plan text; "implementation" reviews the implemented worktree against the plan\'s goals and acceptance criteria (post-execution amelioration).',
41
- }),
42
- ),
43
51
  focus: Type.Optional(Type.String({ description: "Specific concerns to direct the pass at" })),
44
52
  reviewers: Type.Optional(
45
53
  Type.Integer({
46
54
  minimum: 1,
47
55
  maximum: 3,
48
- description: "Number of independent reviewer subagents (big plans: 3 for the concurrent round). Criticizer is always 1.",
56
+ description: "Number of independent reviewer subagents (big plans: 3 for the concurrent round).",
49
57
  }),
50
58
  ),
51
59
  context: Type.Optional(
@@ -54,27 +62,52 @@ const RefineParams = Type.Object({
54
62
  resumeRoundId: Type.Optional(
55
63
  Type.String({
56
64
  description:
57
- 'Round id to resume (I-004). Lanes already complete for this round in the run checkpoint are reused from their persisted outputs; only pending/failed/missing lanes run. Never reuse a round id across plan versions.',
65
+ 'Round id to resume. Lanes already complete for this round in the run checkpoint are reused from their persisted outputs; only pending/failed/missing lanes run. Never reuse a round id across plan versions.',
58
66
  }),
59
67
  ),
60
68
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
61
69
  });
62
70
 
63
- function roleGateError(role: string, roleConfig: RoleConfig | undefined, problem: "mode" | "confirm"): StateError {
71
+ function roleGateError(problem: "mode" | "confirm", guidance?: string): StateError {
64
72
  if (problem === "mode") {
65
73
  return new StateError(
66
- `The ${role} role mode is missing or invalid in .git/pi_plans/config.json. Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session (run the pass yourself in this session) 3. Other 4. Auto-complete — then persist with the plans tool (set-role).`,
74
+ `The reviewer role mode is missing or invalid. Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session (run the pass yourself in this session) 3. Other 4. Auto-complete — then persist with the plans tool (set-role). The reviewer role lives in the global config (${resolveGlobalConfigPath()}).`,
67
75
  );
68
76
  }
69
77
  return new StateError(
70
- `The ${role} model was never confirmed (confirmed_at is null). Ask the model-confirmation question with ask_choice: 1. Inherit the main agent's model (recommended) 2. Choose a model (list options from the /model picker; persist the exact provider/model selector) 3. Other 4. Auto-complete — then persist with the plans tool (set-role, confirmed: true, modelSelector: the selector or 'inherit').`,
78
+ guidance ??
79
+ firstUseTextGuidance([], resolveGlobalConfigPath()),
71
80
  );
72
81
  }
73
82
 
83
+ /** First-use gate shared by the delegated spawn path: pop panels (TUI) or
84
+ * menus (hasUI non-TUI), persist on completion, cancel cleanly on Esc.
85
+ * Returns the role to use for THIS invocation, or throws. */
86
+ async function ensureReviewerReady(
87
+ toolName: string,
88
+ host: RolePanelHost,
89
+ role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
90
+ ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
91
+ if (role.mode === "current-session" || reviewerReady(role as never)) return role as never;
92
+ let outcome: FirstUseOutcome = await runFirstUseFlow(host, role.thinking_level);
93
+ if (outcome.status === "confirmed" && outcome.model_selector !== null) {
94
+ // F-008: validate the freshly chosen selector against the registry when
95
+ // one is present, so a typo'd manual entry fails here, not at spawn.
96
+ if (availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
97
+ outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
98
+ }
99
+ }
100
+ if (outcome.status === "cancelled") throw firstUseCancelledError(toolName);
101
+ if (outcome.status === "unavailable") {
102
+ throw roleGateError("confirm", firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath()));
103
+ }
104
+ if (outcome.role.model_selector === null) throw roleGateError("confirm", firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath()));
105
+ return outcome.role;
106
+ }
107
+
74
108
  function setupRefinementExecution(
75
109
  ctx: ExtensionContext,
76
110
  parentSignal: AbortSignal | undefined,
77
- role: "reviewer" | "criticizer",
78
111
  lanes: Array<{ id: string; label?: string }>,
79
112
  modelLabel?: string,
80
113
  lang: UiLanguage = "en",
@@ -84,7 +117,7 @@ function setupRefinementExecution(
84
117
  if (parentSignal?.aborted) controller.abort();
85
118
  else parentSignal?.addEventListener("abort", relayAbort, { once: true });
86
119
 
87
- const overlay = ctx.mode === "tui" ? new RefineOverlayController(role, lanes, relayAbort, lang) : undefined;
120
+ const overlay = ctx.mode === "tui" ? new RefineOverlayController("reviewer", lanes, relayAbort, lang) : undefined;
88
121
  overlay?.open(refineOverlayContext(ctx), modelLabel);
89
122
 
90
123
  return {
@@ -97,20 +130,20 @@ function setupRefinementExecution(
97
130
  };
98
131
  }
99
132
 
100
- export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
133
+ export function registerRefineTool(ext: ExtensionAPI, baseDir: string): void {
101
134
  const agentsDir = path.join(baseDir, "agents");
102
135
 
103
- const loadAgentPrompt = (role: "reviewer" | "criticizer"): string => {
104
- const file = path.join(agentsDir, `${role}.md`);
136
+ const loadAgentPrompt = (): string => {
137
+ const file = path.join(agentsDir, "reviewer.md");
105
138
  return stripFrontmatter(fs.readFileSync(file, "utf8"));
106
139
  };
107
140
 
108
- pi.registerTool({
141
+ ext.registerTool({
109
142
  name: "refine",
110
143
  label: "Refine",
111
144
  description:
112
- "Run a reviewer or criticizer refinement round on a PLAN_vN.md (target=\"plan\", default) or on the implemented worktree (target=\"implementation\", post-execution amelioration) via read-only Pi subagents. Reviewer: findings with IDs, severity, evidence, impact, fix, disposition. Criticizer: up to five adaptive questions. Use reviewers: 3 for concurrent reviewer rounds (big-plan plan review; implementation-review rounds honor the run's configured reviewerCount when reviewers is omitted). Refuses to spawn until the role's mode and model are confirmed in .git/pi_plans/config.json (ask via ask_choice, persist via the plans tool).",
113
- promptSnippet: "Run reviewer/criticizer plan-refinement rounds",
145
+ "Run a reviewer refinement round on a PLAN_vN.md via read-only Pi subagents. Each reviewer returns findings (F-###, severity, evidence, impact, fix, disposition) AND questions (Q-1..Q-5) that only the user can settle — after the round you MUST ask every question with ask_choice (one call per question or a batched form, in the configured language, stable questionIds) and record the answers before revising the plan. Use reviewers: 3 for concurrent reviewer rounds (big-plan review). Refuses to spawn until the reviewer mode is set and, for delegated-subagent, a concrete model is confirmed: first use pops native model/effort panels in TUI (persisted to the global reviewer config) instead of an ask_choice question.",
146
+ promptSnippet: "Run reviewer plan-refinement rounds",
114
147
  parameters: RefineParams,
115
148
 
116
149
  async execute(_toolCallId, params, signal, _onUpdate, ctx) {
@@ -122,19 +155,23 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
122
155
  throw new StateError("no pi-plans state found; run the plans tool (action: init) first");
123
156
  }
124
157
  const config = loadConfig(root);
125
- const roleConfig = config[params.role] as RoleConfig | undefined;
126
- if (!roleConfig || (roleConfig.mode !== "delegated-subagent" && roleConfig.mode !== "current-session")) {
127
- throw roleGateError(params.role, roleConfig, "mode");
128
- }
129
- if (roleConfig.confirmed_at === null) {
130
- throw roleGateError(params.role, roleConfig, "confirm");
131
- }
132
158
 
133
- // Resolve and read the plan.
159
+ // F-005: cheap validations BEFORE any first-use panel, so a bad planPath
160
+ // never walks the user through two panels that would then be discarded.
134
161
  const planPath = path.resolve(workdir, params.planPath.replace(/^@/, ""));
135
162
  if (!fs.existsSync(planPath)) throw new StateError(`plan file not found: ${planPath}`);
136
163
  const planText = fs.readFileSync(planPath, "utf8");
137
164
 
165
+ // Effective reviewer: global config first, legacy workspace block
166
+ // second — resolved read-only, never written here (F-001).
167
+ const { reviewer: initialRole } = resolveEffectiveReviewer(root);
168
+ if (initialRole.mode !== "delegated-subagent" && initialRole.mode !== "current-session") {
169
+ throw roleGateError("mode");
170
+ }
171
+ // Model/effort confirmation applies only to delegated-subagent
172
+ // (decision 10); current-session runs in this session with its model.
173
+ const roleConfig = await ensureReviewerReady("refine", ctx as unknown as RolePanelHost, initialRole);
174
+
138
175
  const overlayLang = uiLanguageFromTag(config.language.tag);
139
176
  // Record spawns against the active run when one exists.
140
177
  const active = resolveActiveRun(ctx.sessionManager, workdir);
@@ -142,10 +179,10 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
142
179
  if (!active) return;
143
180
  try {
144
181
  recordSubagent(workdir, active.run_id, {
145
- role: params.role,
182
+ role: "reviewer",
146
183
  name,
147
184
  model: model ?? null,
148
- // I-010: meter subagent token/cost for benchmark accounting.
185
+ thinking_level: roleConfig.mode === "current-session" ? null : roleConfig.thinking_level,
149
186
  usage: usage
150
187
  ? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
151
188
  : null,
@@ -155,36 +192,18 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
155
192
  }
156
193
  };
157
194
 
158
- const target = params.target ?? "plan";
159
- // I-004: durable round bookkeeping. Rounds start (or resume) in the
195
+ // Durable round bookkeeping. Rounds start (or resume) in the
160
196
  // checkpoint BEFORE any lane spawns; successful outputs are persisted
161
- // BEFORE the tool result returns (C-007).
197
+ // BEFORE the tool result returns.
162
198
  const checkpointLoad = active ? loadCheckpoint(workdir, active.run_id) : null;
163
199
  const useCheckpoint = checkpointLoad?.status === "ok" ? checkpointLoad.checkpoint : null;
164
- const roundId =
165
- params.resumeRoundId ?? `${target}-${params.role}-r${Date.now().toString(36)}`;
166
- // D-4 durable reviewer-count fallback: an implementation-review reviewer
167
- // round with an omitted `reviewers` reads the run's configured value
168
- // from the checkpoint, so restarts/migrations never silently revert 2/3
169
- // to 1. Explicit params always win; plan-review rounds are unchanged.
170
- const configuredImplReviewers =
171
- target === "implementation" && params.role === "reviewer"
172
- ? useCheckpoint?.implementationReview?.reviewerCount
173
- : undefined;
174
- const roundReviewerCount =
175
- params.role === "reviewer"
176
- ? Math.min(3, Math.max(1, params.reviewers ?? configuredImplReviewers ?? 1))
177
- : 1;
178
- // F-001 (implementation review): the spec MUST carry lanes —
179
- // reviewerLanes(count) for reviewer rounds, one lane for criticizer.
180
- const roundLanes =
181
- params.role === "reviewer"
182
- ? reviewerLanes(roundReviewerCount).map((lane) => ({ laneId: lane.id, lens: lane.lens ?? undefined }))
183
- : [{ laneId: "criticizer" }];
200
+ const roundId = params.resumeRoundId ?? `plan-reviewer-r${Date.now().toString(36)}`;
201
+ const roundReviewerCount = Math.min(3, Math.max(1, params.reviewers ?? 1));
202
+ const roundLanes = reviewerLanes(roundReviewerCount).map((lane) => ({ laneId: lane.id, lens: lane.lens ?? undefined }));
184
203
  const roundSpec = {
185
204
  roundId,
186
- role: params.role,
187
- target,
205
+ role: "reviewer" as const,
206
+ target: "plan" as const,
188
207
  reviewers: roundReviewerCount,
189
208
  planPath,
190
209
  focus: params.focus,
@@ -205,110 +224,43 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
205
224
  /* the subagents ledger still records the spawn; resume treats the lane as unfinished */
206
225
  }
207
226
  };
208
- const pickTask = (role: "reviewer" | "criticizer", lens: string | null): string => {
209
- if (role === "reviewer") {
210
- return target === "implementation"
211
- ? buildImplementationReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context })
212
- : buildReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context });
213
- }
214
- return target === "implementation"
215
- ? buildImplementationCriticizerTask({ planText, planPath, focus: params.focus, context: params.context })
216
- : buildCriticizerTask({ planText, planPath, focus: params.focus, context: params.context });
217
- };
227
+ const pickTask = (lens: string | null): string =>
228
+ buildReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context });
218
229
 
219
- const systemPrompt = loadAgentPrompt(params.role);
230
+ const systemPrompt = loadAgentPrompt();
220
231
  const graphEnabled = config.graph_enabled === true;
221
232
  const subagentTools = graphEnabled ? ["read", "grep", "find", "ls", "code_graph"] : undefined;
222
233
  const graphPrompt = graphBlockForRefiner(graphEnabled);
223
- const inheritModel = ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined;
224
- const model = roleConfig.model_selector ?? inheritModel;
225
- const modelLabel = model ?? "inherit";
234
+ const model = roleConfig.mode === "current-session" ? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined) : roleConfig.model_selector ?? undefined;
235
+ const modelLabel = roleModelLabel(model ?? "inherit", roleConfig.mode === "current-session" ? null : roleConfig.thinking_level);
226
236
 
227
237
  if (roleConfig.mode === "current-session") {
228
- const task = pickTask(params.role, null);
238
+ const task = pickTask(null);
229
239
  return {
230
240
  content: [
231
241
  {
232
242
  type: "text",
233
- text: `Role mode is current-session: perform the read-only ${params.role} pass yourself, in this session, following this brief. Do not spawn anything.\n\n${task}`,
243
+ text: `Role mode is current-session: perform the read-only reviewer pass yourself, in this session, following this brief. Do not spawn anything. Then surface the findings and ask the Questions section with ask_choice before revising.\n\n${task}`,
234
244
  },
235
245
  ],
236
- details: { mode: "current-session", role: params.role, planPath, target },
246
+ details: { mode: "current-session", planPath },
237
247
  };
238
248
  }
239
249
 
240
- if (params.role === "criticizer") {
241
- const laneId = "criticizer";
242
- const persisted = reusable[laneId];
243
- if (persisted) {
244
- return {
245
- content: [
246
- {
247
- type: "text",
248
- text: `${readReviewOutput(workdir, active!.run_id, persisted)}\n\n---\nReused the persisted criticizer result for round ${roundId} (no re-run). Ask each criticizer question with ask_choice (one call per question, in the configured language), record every answer, then revise the plan only after every question has an answer.`,
249
- },
250
- ],
251
- details: { mode: "delegated-subagent", role: params.role, planPath, target, roundId, reused: true },
252
- };
253
- }
254
- const name = `${roleConfig.name_prefix}-criticizer-${Date.now().toString(36)}`;
255
- const execution = setupRefinementExecution(ctx, signal, "criticizer", [{ id: name, label: "criticizer" }], modelLabel, overlayLang);
256
- try {
257
- const result = await runPiSubagent({
258
- systemPrompt: `${systemPrompt}\n\n${graphPrompt}`,
259
- task: pickTask("criticizer", null),
260
- cwd: workdir,
261
- model,
262
- tools: subagentTools,
263
- signal: execution.signal,
264
- onProgress: (event) => execution.overlay?.update(name, event),
265
- });
266
- execution.overlay?.complete(name, result);
267
- record(name, result.ok ? result.model ?? model : null, result.usage);
268
- persistOutcome(laneId, result.ok ? { ok: true, output: result.output } : { ok: false, error: result.errorMessage });
269
- if (!result.ok) {
270
- throw new Error(
271
- `criticizer subagent failed: ${result.errorMessage ?? "unknown error"}${result.stderr ? `\nstderr: ${result.stderr.slice(0, 2000)}` : ""}`,
272
- );
273
- }
274
- return {
275
- content: [
276
- {
277
- type: "text",
278
- text: `${result.output}\n\n---\nAsk each criticizer question with ask_choice (one call per question, in the configured language, with a stable questionId per question), record every answer, then revise the plan only after every question has an answer. After the revision, record the boundary: plans record-checkpoint (checkpoint: { transition: "review-consolidated", roundId: "${roundId}" }).`,
279
- },
280
- ],
281
- details: { mode: "delegated-subagent", role: params.role, planPath, target, roundId, model: result.model ?? model },
282
- };
283
- } finally {
284
- await execution.close();
285
- }
286
- }
287
-
288
- const count = Math.min(3, Math.max(1, params.reviewers ?? configuredImplReviewers ?? 1));
250
+ const count = roundReviewerCount;
289
251
  const lanes = reviewerLanes(count);
290
252
  const jobs = lanes.map((lane) => {
291
253
  const name = `${roleConfig.name_prefix}-${active?.run_id ?? "adhoc"}-${lane.id}`;
292
- const task = pickTask("reviewer", lane.lens);
254
+ const task = pickTask(lane.lens);
293
255
  return { lane, name, task };
294
256
  });
295
257
 
296
- // Per-plan amelioration round counter (post-execution loop auditability).
297
- const roundsSlot = ctx.sessionManager as unknown as { __ameliorateRounds?: Map<string, number> };
298
- const nextRound = (planPath: string): number => {
299
- roundsSlot.__ameliorateRounds ??= new Map();
300
- const round = (roundsSlot.__ameliorateRounds.get(planPath) ?? 0) + 1;
301
- roundsSlot.__ameliorateRounds.set(planPath, round);
302
- return round;
303
- };
304
-
305
- // Lane-level resume (F-007): completed lanes are reused from their
258
+ // Lane-level resume: completed lanes are reused from their
306
259
  // persisted outputs; only pending/failed/missing lanes spawn.
307
260
  const runnableJobs = jobs.filter((job) => reusable[job.lane.id] === undefined);
308
261
  const execution = setupRefinementExecution(
309
262
  ctx,
310
263
  signal,
311
- "reviewer",
312
264
  runnableJobs.map((job) => ({ id: job.lane.id, label: job.lane.id })),
313
265
  modelLabel,
314
266
  overlayLang,
@@ -322,27 +274,16 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
322
274
  task: job.task,
323
275
  cwd: workdir,
324
276
  model,
277
+ thinkingLevel: roleConfig.thinking_level ?? undefined,
325
278
  tools: subagentTools,
326
279
  signal: execution.signal,
327
280
  onProgress: (event) => execution.overlay?.update(job.lane.id, event),
328
281
  });
329
282
  execution.overlay?.complete(job.lane.id, result);
330
283
  record(job.name, result.ok ? result.model ?? model : null, result.usage);
331
- // Persist BEFORE returning (C-007): a crash after this point
284
+ // Persist BEFORE returning: a crash after this point
332
285
  // still leaves the lane reusable.
333
286
  persistOutcome(job.lane.id, result.ok ? { ok: true, output: result.output } : { ok: false, error: result.errorMessage });
334
- if (target === "implementation" && result.ok) {
335
- try {
336
- pi.appendEntry("pi-plans-ameliorate", {
337
- planPath,
338
- phase: "round",
339
- currentRound: nextRound(planPath),
340
- lane: job.lane.id,
341
- });
342
- } catch {
343
- /* appendEntry is best-effort; audit trail survives in subagents.jsonl */
344
- }
345
- }
346
287
  return { job, result };
347
288
  } catch (error) {
348
289
  record(job.name, null);
@@ -385,7 +326,7 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
385
326
  if (failures === results.length && reusedCount === 0) {
386
327
  const first = results[0];
387
328
  throw new Error(
388
- `all reviewer subagents failed: ${first?.result.errorMessage ?? "unknown error"}${first?.result.stderr ? `\nstderr: ${first.result.stderr.slice(0, 2000)}` : ""}${model ? `\nIf the model selector "${model}" is unavailable, reset the confirmation (plans set-role --reset-confirmation) and re-ask the model-confirmation question.` : ""}`,
329
+ `all reviewer subagents failed: ${first?.result.errorMessage ?? "unknown error"}${first?.result.stderr ? `\nstderr: ${first.result.stderr.slice(0, 2000)}` : ""}${model ? `\nIf the model selector "${model}" is unavailable, reset the confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next refine opens the native model panel to re-confirm.` : ""}`,
389
330
  );
390
331
  }
391
332
 
@@ -398,17 +339,17 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
398
339
  content: [
399
340
  {
400
341
  type: "text",
401
- text: `${text}\n\n---\nConsolidate: merge and dedupe findings into PLAN_vN_reviewer_comments.md${count > 1 ? " (one consolidated file; keep each finding's source reviewer, severity, evidence, and disposition)" : ""}, accept or reject each finding on repo/reference evidence, surface at most five high-priority findings to the user, then immediately ask the next refinement-mode question with ask_choice. Then record the boundary: plans record-checkpoint (checkpoint: { transition: "review-consolidated", roundId: "${roundId}", dispositionArtifact: "<comments file, run-dir relative>" }).${target === "implementation" ? ' When the whole round is disposed, also record (checkpoint: { transition: "implementation-round-finished" }); when the termination condition is met, close with (checkpoint: { transition: "completed", evidence: "<why the condition is satisfied>" }).' : ""}`,
342
+ text: `${text}\n\n---\nConsolidate: merge and dedupe findings into PLAN_vN_reviewer_comments.md${count > 1 ? " (one consolidated file; keep each finding's source reviewer, severity, evidence, and disposition)" : ""}, accept or reject each finding on repo/reference evidence, merge the Questions sections into one deduped list, surface at most five high-priority findings to the user — then ask EVERY consolidated question with ask_choice (batch them into one questions:[...] form or ask one per call, in the configured language, with stable questionIds), record every answer, and only then revise the plan. After the revision, record the boundary: plans record-checkpoint (checkpoint: { transition: "review-consolidated", roundId: "${roundId}", dispositionArtifact: "<comments file, run-dir relative>" }).`,
402
343
  },
403
344
  ],
404
345
  details: {
405
346
  mode: "delegated-subagent",
406
- role: "reviewer",
407
347
  planPath,
408
348
  roundId,
409
349
  reusedLanes: Object.keys(reusable),
410
350
  reviewers: count,
411
351
  model,
352
+ thinkingLevel: roleConfig.thinking_level,
412
353
  outputs: results.map(({ job, result }) => ({ name: job.name, lane: job.lane.id, lens: job.lane.lens, ok: result.ok, output: result.output, stderr: result.stderr, turns: result.turns })),
413
354
  },
414
355
  };
@@ -418,14 +359,10 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
418
359
  },
419
360
 
420
361
  renderCall(args, theme) {
421
- const count = args.role === "reviewer" ? args.reviewers ?? 1 : 1;
422
- let text =
423
- theme.fg("toolTitle", theme.bold("refine ")) +
424
- theme.fg("accent", args.role) +
425
- theme.fg("muted", count > 1 ? ` ×${count}` : "");
362
+ const count = args.reviewers ?? 1;
363
+ let text = theme.fg("toolTitle", theme.bold("refine ")) + theme.fg("accent", "reviewer") + theme.fg("muted", count > 1 ? ` ×${count}` : "");
426
364
  const short = args.planPath ? args.planPath.split("/").pop() : "";
427
365
  if (short) text += theme.fg("dim", ` ${short}`);
428
- if (args.target === "implementation") text += theme.fg("dim", " (implementation)");
429
366
  if (args.focus) text += `\n${theme.fg("dim", ` focus: ${args.focus.slice(0, 80)}`)}`;
430
367
  return new Text(text, 0, 0);
431
368
  },
@@ -1,18 +0,0 @@
1
- ---
2
- name: pi-plans-criticizer
3
- description: Read-only adversarial questioner for pi-plans refinement rounds; stress-tests plan assumptions with adaptive questions.
4
- tools: read, grep, find, ls
5
- ---
6
-
7
- You are a read-only criticizer in the pi-plans workflow.
8
-
9
- Rules:
10
-
11
- - Perform read-only analysis. Never edit, write, or delete any file.
12
- - Stress-test the plan's assumptions; do not rewrite the plan.
13
- - You may inspect the repository to ground your questions.
14
-
15
- Output Markdown in exactly this shape:
16
-
17
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
18
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be concrete and answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
@@ -1,26 +0,0 @@
1
- ---
2
- name: pi-plans-executor
3
- description: Delegated plan executor for pi-plans runs; autonomously implements an accepted plan end-to-end in the target workdir and reports Verifier-Checklist progress with markers.
4
- tools: read, write, edit, bash, grep, find, ls
5
- ---
6
-
7
- You are a delegated plan executor in the pi-plans workflow. The parent session handed you an accepted plan; you implement it completely and autonomously.
8
-
9
- Rules:
10
-
11
- - Work autonomously. You have NO user to ask questions — the ask_choice tool is unavailable in this context. When a decision is genuinely ambiguous, choose the option most consistent with the plan's goals and constraints and record the deviation in your final summary.
12
- - Implement the plan file you were given, in dependency order. Read the plan first; it is the single source of truth for scope, requirements, and verification steps.
13
- - Write code directly. Your `write`/`edit` tools operate natively on disk (any DB-first staging in the parent workdir is bypassed for you); no `apply` step is needed.
14
- - Follow the plan's own execution rules: smallest end-to-end slice first, then layer; no speculative abstractions; no backward-compatibility fallbacks; prefer established libraries already in the project.
15
- - Emit progress markers IN YOUR REPLIES as you go: `[DONE:VC-xxx]` once a verifier item's stated evidence passes, `[I-###:implemented]` / `[I-###:validating]` for implementation items when the plan defines them. The parent session parses these markers from your streamed messages to update the tracked checklist — put them in message text, not only in the final output.
16
- - Run the plan's verification steps yourself (tests, validate scripts) and only mark a VC done when its stated evidence actually passes.
17
- - Do not modify pi-plans state (run.json, checkpoints, ledgers) — the parent owns the run bookkeeping.
18
- - If a verification step is impossible in this environment, leave the VC unmarked and explain in the summary.
19
-
20
- Finish with a structured summary in exactly this shape:
21
-
22
- - Completed VCs: <ids or none>
23
- - Remaining VCs: <ids or none, with one-line reasons>
24
- - Implementation items: <per-item state>
25
- - Deviations from the plan: <any decisions you made on ambiguous points>
26
- - Evidence: <commands run and their results>