pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
@@ -7,14 +7,11 @@ import * as os from "node:os";
7
7
  import * as path from "node:path";
8
8
  import { after, before, describe, it } from "node:test";
9
9
  import {
10
- applyCompleted,
11
10
  applyExecutionApproved,
12
11
  applyExecutionCompleted,
13
12
  applyExecutionProgress,
14
13
  applyExecutionHeadChanged,
15
14
  applyExecutionStopped,
16
- applyImplementationReviewConfigured,
17
- applyImplementationRoundFinished,
18
15
  applyLaneResult,
19
16
  applyMigration,
20
17
  applyPlanWritten,
@@ -331,33 +328,21 @@ describe("state machine reducers", () => {
331
328
  assert.throws(() => applyLaneResult(staged, "r2", "l1", { ok: true }), StateError);
332
329
  });
333
330
 
334
- it("implementation review: condition first, rounds counted, completed needs evidence (F-004)", () => {
335
- const { workdir, runId } = setupRun("sm-impl");
331
+ it("implementation-review write side is gone; legacy checkpoints stay readable (D-018)", () => {
332
+ const { workdir, runId } = setupRun("sm-impl-legacy");
336
333
  createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
337
334
  let cp = baseCheckpoint(workdir, runId);
338
- const planPath = path.join(path.dirname(checkpointFilePath(workdir, runId)!), "PLAN_v1.md");
339
- fs.writeFileSync(planPath, "# plan", "utf8");
340
- const plan = planIdentityOf(planPath, 1);
341
- cp = { ...cp, plan, phase: "implementation-review", nextAction: "ask-question" };
342
- assert.throws(
343
- () => applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] }),
344
- StateError,
345
- );
346
- cp = applyImplementationReviewConfigured(cp, "until-no-high");
347
- assert.throws(() => applyImplementationReviewConfigured(cp, "again"), StateError);
348
- cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] });
349
- const file = writeReviewOutput(workdir, runId, "i1", "l1", "out");
350
- cp = applyLaneResult(cp, "i1", "l1", { ok: true, resultFile: file });
351
- assert.throws(() => applyImplementationRoundFinished(cp), StateError);
352
- cp = applyReviewConsolidated(cp, "i1");
353
- cp = applyImplementationRoundFinished(cp);
335
+ // Simulate a persisted 0.6.0 checkpoint in the legacy phase with its
336
+ // review bookkeeping — the schema still parses it read-only.
337
+ cp = {
338
+ ...cp,
339
+ phase: "implementation-review",
340
+ nextAction: "run-review",
341
+ implementationReview: { terminationCondition: "until-no-high", reviewerCount: 2, completedRounds: 1 },
342
+ };
343
+ const round = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 2, lanes: [{ laneId: "l1" }] });
344
+ assert.equal(round.reviewRounds.length, 1);
354
345
  assert.equal(cp.implementationReview?.completedRounds, 1);
355
- // Completion requires evidence AND a termination condition.
356
- assert.throws(() => applyCompleted({ ...cp, implementationReview: undefined }, "evidence"), StateError);
357
- assert.throws(() => applyCompleted(cp, " "), StateError);
358
- const done = applyCompleted(cp, "no findings in final round");
359
- assert.equal(done.phase, "completed");
360
- assert.equal(done.nextAction, "none");
361
346
  });
362
347
 
363
348
  it("execution approval and progress (D-003/D-011)", () => {
@@ -394,10 +379,12 @@ describe("state machine reducers", () => {
394
379
  assert.equal(cp.execution?.pausedReason, "stopped by user");
395
380
  const resumed = applyExecutionProgress(cp, { pausedReason: null });
396
381
  assert.equal(resumed.execution?.pausedReason, undefined);
397
- // Completion hands over to the implementation-review phase (R-005).
382
+ // Completion: v0.6.1 passes a completed audit straight to the terminal
383
+ // phase (the implementation-review loop is gone, D-018).
398
384
  const finished = applyExecutionCompleted(cp);
399
- assert.equal(finished.phase, "implementation-review");
400
- assert.equal(finished.nextAction, "ask-question");
385
+ assert.equal(finished.phase, "completed");
386
+ assert.equal(finished.nextAction, "none");
387
+ assert.equal(finished.execution?.audit?.passed, true);
401
388
  });
402
389
 
403
390
  it("migration resets rounds and approval, keeps termination (F-003)", () => {
@@ -431,72 +418,3 @@ describe("state machine reducers", () => {
431
418
  });
432
419
  });
433
420
 
434
- describe("implementationReview.reviewerCount (0.5.4)", () => {
435
- it("configured persists reviewerCount and survives checkpoint roundtrip", () => {
436
- const { workdir, runId } = setupRun("rc-persist");
437
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
438
- let cp = baseCheckpoint(workdir, runId);
439
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
440
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 2);
441
- assert.equal(cp.implementationReview?.reviewerCount, 2);
442
- assert.equal(cp.nextAction, "run-review");
443
- mutateCheckpoint(workdir, runId, () => cp);
444
- const reloaded = loadCheckpoint(workdir, runId);
445
- assert.ok(reloaded.status === "ok");
446
- assert.equal(reloaded.checkpoint.implementationReview?.reviewerCount, 2);
447
- });
448
-
449
- it("omitted reviewerCount stays undefined (legacy checkpoints unchanged)", () => {
450
- const { workdir, runId } = setupRun("rc-legacy");
451
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
452
- let cp = baseCheckpoint(workdir, runId);
453
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
454
- cp = applyImplementationReviewConfigured(cp, "1 round");
455
- assert.equal(cp.implementationReview?.reviewerCount, undefined);
456
- mutateCheckpoint(workdir, runId, () => cp);
457
- assert.ok(loadCheckpoint(workdir, runId).status === "ok");
458
- });
459
-
460
- it("rejects out-of-range and non-integer reviewerCount on load", () => {
461
- const { workdir, runId } = setupRun("rc-invalid");
462
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
463
- const file = checkpointFilePath(workdir, runId)!;
464
- const base = JSON.parse(fs.readFileSync(file, "utf8")) as { implementationReview?: unknown };
465
- for (const bad of [0, 4, "3", 1.5]) {
466
- const doc = {
467
- ...base,
468
- implementationReview: {
469
- terminationCondition: "1 round",
470
- reviewerCount: bad,
471
- completedRounds: 0,
472
- },
473
- };
474
- fs.writeFileSync(file, JSON.stringify(doc), "utf8");
475
- const loaded = loadCheckpoint(workdir, runId);
476
- assert.ok(loaded.status === "corrupt", `reviewerCount ${JSON.stringify(bad)} rejected`);
477
- }
478
- // Corrupt bytes refuse overwrite (mutateCheckpoint guard), which is the
479
- // intended fail-loud behavior — no restore attempted here.
480
- });
481
-
482
- it("applyMigration preserves an explicit reviewerCount (CQ1/D-4)", () => {
483
- const { workdir, runId } = setupRun("rc-migrate");
484
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
485
- let cp = baseCheckpoint(workdir, runId);
486
- cp = { ...cp, phase: "implementation-review", nextAction: "run-review" };
487
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
488
- cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 3, lanes: [{ laneId: "l1" }] });
489
- const migrated = applyMigration(cp, { workdir: "/target/wt", worktreeRoot: "/target/wt", commonDir: "/target/.git" });
490
- assert.equal(migrated.implementationReview?.reviewerCount, 3);
491
- assert.equal(migrated.implementationReview?.completedRounds, 0);
492
- });
493
-
494
- it("second configuration write is rejected even with identical values (replay guard)", () => {
495
- const { workdir, runId } = setupRun("rc-replay");
496
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
497
- let cp = baseCheckpoint(workdir, runId);
498
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
499
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
500
- assert.throws(() => applyImplementationReviewConfigured(cp, "until-no-high", 3), StateError);
501
- });
502
- });
@@ -1,9 +1,12 @@
1
1
  /**
2
2
  * `analyze_refs` tool — plan-with-refs per-reference analysis via read-only Pi
3
3
  * subagents with isolated context. One lane per reference (cwd = the ref's own
4
- * directory), reusing the reviewer role gates from `.git/pi_plans/config.json`
5
- * and the concurrent refinement overlay (title "Refs"). Batches are capped at
6
- * three concurrent lanes; larger ref sets run as sequential batches.
4
+ * directory), reusing the reviewer model confirmation from the GLOBAL config
5
+ * (`~/.pi/pi-plans/config.json`) and the concurrent overlay (title "Refs").
6
+ * analyze_refs is spawn-only by nature, so the reviewer MODE is deliberately
7
+ * not consulted here (Q-4=B): a current-session reviewer still gets spawned
8
+ * ref-analyst lanes, with a one-time notice in the result. Batches are capped
9
+ * at three concurrent lanes; larger ref sets run as sequential batches.
7
10
  *
8
11
  * Recording is best-effort: spawns land in `subagents.jsonl` (role
9
12
  * `ref-analyst`) only when an active planning run exists. Analysis output is
@@ -17,7 +20,18 @@ import { Text } from "@earendil-works/pi-tui";
17
20
  import { Type } from "typebox";
18
21
  import * as fs from "node:fs";
19
22
  import * as path from "node:path";
20
- import { loadConfig, normalizeWorkdir, readActive, recordSubagent, resolveStateRootOrNull, StateError } from "../src/state.ts";
23
+ import {
24
+ loadConfig,
25
+ normalizeWorkdir,
26
+ readActive,
27
+ recordSubagent,
28
+ resolveEffectiveReviewer,
29
+ resolveGlobalConfigPath,
30
+ resolveStateRootOrNull,
31
+ StateError,
32
+ } from "../src/state.ts";
33
+ import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type RolePanelHost } from "../src/role-panels.ts";
34
+ import { roleModelLabel } from "../src/thinking-levels.ts";
21
35
  import type { SubagentUsage } from "../src/subagent.ts";
22
36
  import { resolveActiveRun } from "../src/run-context.ts";
23
37
  import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
@@ -45,25 +59,39 @@ const AnalyzeRefsParams = Type.Object({
45
59
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
46
60
  });
47
61
 
48
- function gateError(problem: "state" | "mode" | "current-session" | "confirm"): StateError {
62
+ function gateError(problem: "state" | "confirm", guidance?: string): StateError {
49
63
  if (problem === "state") {
50
64
  return new StateError("no pi-plans state found; run the plans tool (action: init) first");
51
65
  }
52
- if (problem === "mode") {
53
- return new StateError(
54
- "The reviewer role mode is missing or invalid in .git/pi_plans/config.json (analyze_refs reuses the reviewer gates). Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer).",
55
- );
56
- }
57
- if (problem === "current-session") {
58
- return new StateError(
59
- "The reviewer role mode is current-session, but analyze_refs only spawns delegated read-only subagents (one per reference). Ask the user to switch the reviewer mode to delegated-subagent via ask_choice, persist with the plans tool (set-role, role=reviewer, mode=delegated-subagent), then retry analyze_refs.",
60
- );
61
- }
62
66
  return new StateError(
63
- "The reviewer model was never confirmed (confirmed_at is null); analyze_refs reuses the reviewer confirmation. Ask the model-confirmation question with ask_choice: 1. Inherit the main agent's model (recommended) 2. Choose a model (list options from the /model picker; persist the exact provider/model selector) 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer, confirmed: true, modelSelector: the selector or 'inherit').",
67
+ guidance ??
68
+ firstUseTextGuidance([], resolveGlobalConfigPath()),
64
69
  );
65
70
  }
66
71
 
72
+ /** First-use model confirmation for the spawn-only ref-analyst path: native
73
+ * panels in TUI, menus for hasUI non-TUI, embedded text guidance otherwise.
74
+ * The reviewer MODE is not consulted (Q-4=B), but because analysis always
75
+ * spawns, a confirmed CONCRETE model is required even when the stored mode
76
+ * is current-session (model confirmation “as usual”). */
77
+ async function ensureRefAnalystModelReady(
78
+ host: RolePanelHost,
79
+ role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
80
+ ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
81
+ if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
82
+ let outcome = await runFirstUseFlow(host, role.thinking_level);
83
+ if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
84
+ // F-008: a manually entered selector that the registry does not know —
85
+ // one re-pick, then let spawn-side errors surface precisely.
86
+ outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
87
+ }
88
+ if (outcome.status === "cancelled") throw firstUseCancelledError("analyze_refs");
89
+ const guidance = firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath());
90
+ if (outcome.status === "unavailable") throw gateError("confirm", guidance);
91
+ if (outcome.role.model_selector === null) throw gateError("confirm", guidance);
92
+ return outcome.role;
93
+ }
94
+
67
95
  interface AnalysisJob {
68
96
  input: RefAnalystTaskInput;
69
97
  name: string;
@@ -72,14 +100,14 @@ interface AnalysisJob {
72
100
  missing: string | null;
73
101
  }
74
102
 
75
- export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void {
103
+ export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): void {
76
104
  const agentPrompt = stripFrontmatter(fs.readFileSync(path.join(baseDir, "agents", "ref-analyst.md"), "utf8"));
77
105
 
78
- pi.registerTool({
106
+ ext.registerTool({
79
107
  name: "analyze_refs",
80
108
  label: "Analyze Refs",
81
109
  description:
82
- "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer role gates and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. Refuses until the reviewer mode/model gates pass in .git/pi_plans/config.json.",
110
+ "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. The reviewer mode is not consulted (spawn-only); first use pops native model/effort panels in TUI.",
83
111
  promptSnippet: "Analyze plan-with-refs references with per-ref read-only subagents",
84
112
  parameters: AnalyzeRefsParams,
85
113
 
@@ -92,17 +120,22 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
92
120
  throw gateError("state");
93
121
  }
94
122
  const config = loadConfig(root);
95
- const reviewer = config.reviewer;
96
- if (!reviewer || (reviewer.mode !== "delegated-subagent" && reviewer.mode !== "current-session")) {
97
- throw gateError("mode");
98
- }
99
- if (reviewer.mode === "current-session") {
100
- throw gateError("current-session");
101
- }
102
- if (reviewer.confirmed_at === null) {
103
- throw gateError("confirm");
123
+
124
+ // F-005: cheap validations BEFORE any first-use panel.
125
+ if (params.refs.length === 0) {
126
+ throw new StateError("analyze_refs requires at least one reference");
104
127
  }
105
128
 
129
+ // Effective reviewer from the global config (mode NOT consulted —
130
+ // analyze_refs is spawn-only, Q-4=B; a notice surfaces when the stored
131
+ // mode is current-session so the switch is never silent).
132
+ const { reviewer: initialReviewer } = resolveEffectiveReviewer(root);
133
+ const modeIgnoredNotice =
134
+ initialReviewer.mode === "current-session"
135
+ ? `note: the reviewer mode is ${initialReviewer.mode}, but analyze_refs always spawns read-only subagents; the mode is ignored here and unchanged.`
136
+ : null;
137
+ const reviewer = await ensureRefAnalystModelReady(ctx as unknown as RolePanelHost, initialReviewer);
138
+
106
139
  // Resolve refs and validate directories up front; missing ones become
107
140
  // FAILED sections instead of aborting the whole batch.
108
141
  const active = resolveActiveRun(ctx.sessionManager, workdir);
@@ -122,7 +155,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
122
155
  }
123
156
 
124
157
  const model = reviewer.model_selector ?? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined);
125
- const modelLabel = model ?? "inherit";
158
+ const modelLabel = roleModelLabel(model ?? "inherit", reviewer.thinking_level);
126
159
  let languageTag: string | null = null;
127
160
  if (active) {
128
161
  try {
@@ -140,6 +173,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
140
173
  role: "ref-analyst",
141
174
  name,
142
175
  model: okModel ?? model ?? null,
176
+ thinking_level: reviewer.thinking_level,
143
177
  // I-010: meter subagent token/cost for benchmark accounting.
144
178
  usage: usage
145
179
  ? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
@@ -160,6 +194,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
160
194
  task: buildRefAnalystTask({ ...job.input, languageTag }),
161
195
  cwd: job.dir,
162
196
  model,
197
+ thinkingLevel: reviewer.thinking_level ?? undefined,
163
198
  tools: READ_ONLY_TOOLS,
164
199
  signal: relay.signal,
165
200
  onProgress: (event) => overlay?.update(job.laneId, event),
@@ -224,7 +259,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
224
259
 
225
260
  if (failures === jobs.length) {
226
261
  throw new Error(
227
- `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) and re-ask the model-confirmation question.` : ""}`,
262
+ `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next analyze_refs opens the native model panel to re-confirm.` : ""}`,
228
263
  );
229
264
  }
230
265
 
@@ -237,13 +272,13 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
237
272
  content: [
238
273
  {
239
274
  type: "text",
240
- text: `${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
275
+ text: `${modeIgnoredNotice ? `${modeIgnoredNotice}\n\n` : ""}${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
241
276
  },
242
277
  ],
243
278
  details: {
244
279
  mode: "delegated-subagent",
245
280
  role: "ref-analyst",
246
- reviewerGates: { mode: reviewer.mode, model },
281
+ reviewerGates: { mode: initialReviewer.mode, modeIgnored: modeIgnoredNotice !== null, model, thinkingLevel: reviewer.thinking_level },
247
282
  batches: Math.ceil(jobs.length / BATCH_SIZE),
248
283
  model,
249
284
  outputs,
@@ -16,13 +16,6 @@ import { Text } from "@earendil-works/pi-tui";
16
16
  import { Type } from "typebox";
17
17
  import { disableAutoComplete, enableAutoComplete, isAutoCompleteEnabled, recordAskChoice } from "../src/autocomplete.ts";
18
18
  import { assertAutoApprovable, isAutoApproveEnabled } from "../src/auto-approve.ts";
19
- import {
20
- TERMINATION_QUESTION,
21
- TERMINATION_OPTIONS,
22
- TERMINATION_RECORDING_INSTRUCTIONS,
23
- implReviewerCountPromptLine,
24
- renderTerminationOptions,
25
- } from "../src/termination-prompt.ts";
26
19
  import { truncateToWidth, visibleWidth } from "../src/refine-ui-helpers.ts";
27
20
  import { stripRecommendedMarker } from "../src/ask-form.ts";
28
21
  import {
@@ -63,8 +56,8 @@ export const FALLBACK_ROWS = 30;
63
56
  /** Minimal-form floor for tiny terminals (stage-3 width). */
64
57
  const MINIMAL_LINE_WIDTH = 20;
65
58
  /**
66
- * Truncation floor for fixed tail labels (Other…/Auto-complete/Auto-refine
67
- * loop): the longest magic prefix ("Auto-refine loop", 16 cols) plus slack.
59
+ * Truncation floor for fixed tail labels (Other…/Auto-complete): the
60
+ * longest magic prefix ("Auto-complete", 14 cols) plus slack.
68
61
  * These labels drive startsWith() answer routing and must never lose it.
69
62
  */
70
63
  const FIXED_LABEL_FLOOR = 18;
@@ -74,7 +67,7 @@ export interface PanelItem {
74
67
  core: string;
75
68
  /** Full display label: core + description (degradation stage 0). */
76
69
  display: string;
77
- /** Fixed tail labels (Other…/Auto-complete/Auto-refine loop): truncation keeps at least the magic prefix. */
70
+ /** Fixed tail labels (Other…/Auto-complete): truncation keeps at least the magic prefix. */
78
71
  fixed?: boolean;
79
72
  }
80
73
 
@@ -213,12 +206,6 @@ export const AskChoiceParams = Type.Object(
213
206
  purpose: Type.Optional(
214
207
  Type.String({ description: "Short machine-readable purpose (e.g. 'scope', 'termination-condition')." }),
215
208
  ),
216
- trailing: Type.Optional(
217
- StringEnum(["auto-refine-loop"] as const, {
218
- description:
219
- 'Replace the trailing Auto-complete option with "Auto-refine loop" (post-execution amelioration prompt). Selecting it returns instructions to ask the rounds/termination follow-up; Auto-complete is suppressed entirely for this question.',
220
- }),
221
- ),
222
209
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
223
210
  },
224
211
  { additionalProperties: false },
@@ -589,12 +576,12 @@ function formatBatchAnswers(batch: NonNullable<AskChoiceDetails["batch"]>): stri
589
576
  }
590
577
  const NL = "\n";
591
578
 
592
- export function registerAskChoiceTool(pi: ExtensionAPI): void {
593
- pi.registerTool({
579
+ export function registerAskChoiceTool(ext: ExtensionAPI): void {
580
+ ext.registerTool({
594
581
  name: "ask_choice",
595
582
  label: "Ask Choice",
596
583
  description:
597
- "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt. EVERY option you author — including the accept/execute handoff and the implementation-review setup questions — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
584
+ "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the questionIds reserved for handoff. EVERY option you author — including the accept/execute handoff — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
598
585
  promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
599
586
  promptGuidelines: [
600
587
  "Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
@@ -605,13 +592,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
605
592
  executionMode: "sequential",
606
593
 
607
594
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
608
- // R-13 (defense-in-depth): a delegated executor child has no user to
609
- // answer — refuse instead of blocking a headless run on a UI prompt.
610
- if (process.env.PI_PLANS_EXECUTOR === "1") {
611
- throw new Error(
612
- "ask_choice is unavailable in a delegated executor session: no interactive user. Decide autonomously, proceed, and record the deviation in your final summary.",
613
- );
614
- }
615
595
  // 0.4.0 batch mode: one tabbed form for a whole round of questions
616
596
  // (2-8). The single-question path below is untouched (C-004).
617
597
  // F-005 (impl review r1): ambiguous shapes fail loudly instead of
@@ -620,9 +600,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
620
600
  if (params.question !== undefined || params.options !== undefined) {
621
601
  throw new Error("ask_choice accepts either question+options or questions, not both");
622
602
  }
623
- if (params.trailing !== undefined) {
624
- throw new Error("ask_choice batch mode does not support trailing (single-question only)");
625
- }
626
603
  return executeAskChoiceBatch({ questions: params.questions, workdir: params.workdir }, ctx);
627
604
  }
628
605
  const workdir = normalizeWorkdir(params.workdir ?? ctx.cwd);
@@ -661,10 +638,7 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
661
638
  /* the decisions ledger already holds the answer; F-005 reconcile covers the gap */
662
639
  }
663
640
  };
664
- // Param normalization: a trailing option replaces Auto-complete entirely,
665
- // so an erroneously passed autoComplete flag is suppressed here.
666
- const trailing = params.trailing;
667
- const autoComplete = (params.autoComplete ?? true) && trailing === undefined;
641
+ const autoComplete = params.autoComplete ?? true;
668
642
  const options = params.options;
669
643
  if (options.length === 0) throw new Error("ask_choice requires at least one option");
670
644
  const recommended = options.find((option) => option.recommended) ?? options[0];
@@ -764,8 +738,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
764
738
  };
765
739
  }
766
740
 
767
- const AUTO_REFINE_LOOP_LABEL =
768
- "Auto-refine loop (run refinement rounds until no high-severity finding or the 5-round cap)";
769
741
  const panelItems: PanelItem[] = options.map((option, index) => {
770
742
  const label = stripRecommendedMarker(option.label);
771
743
  const isRec = option === recommended;
@@ -777,7 +749,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
777
749
  });
778
750
  if (allowOther) panelItems.push({ core: "Other… (type your own answer)", display: "Other… (type your own answer)", fixed: true });
779
751
  if (autoComplete) panelItems.push({ core: "Auto-complete (take the recommended option)", display: "Auto-complete (take the recommended option)", fixed: true });
780
- else if (trailing) panelItems.push({ core: AUTO_REFINE_LOOP_LABEL, display: AUTO_REFINE_LOOP_LABEL, fixed: true });
781
752
 
782
753
  const panel = fitAskChoicePanel(
783
754
  params.question,
@@ -820,23 +791,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
820
791
  };
821
792
  }
822
793
 
823
- if (trailing && selected.startsWith("Auto-refine loop")) {
824
- recordAskChoice(ctx, false);
825
- record("Auto-refine loop", "user");
826
- // Skill-aware reviewer-count default (D-1/D-4): same mapping the
827
- // goal-running continuation in src/exec.ts renders.
828
- const activeSkill = resolveActiveRun(ctx.sessionManager, workdir)?.skill;
829
- return {
830
- content: [
831
- {
832
- type: "text",
833
- text: `User selected Auto-refine loop. Immediately ask the follow-up with ask_choice (autoComplete: false, in the session language): "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(activeSkill)} Then run the loop per the completion instructions: each round calls refine (role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>), accepts findings on evidence, applies fixes, re-runs relevant tests, and continues until the chosen termination condition — the goal-wait option keeps the loop running until no unpassed VCs remain.`,
834
- },
835
- ],
836
- details: details("Auto-refine loop", "user"),
837
- };
838
- }
839
-
840
794
  if (allowOther && selected.startsWith("Other…")) {
841
795
  const typed = await ctx.ui.input(`${params.question} — your answer:`);
842
796
  if (typed === undefined || !typed.trim()) {
@@ -128,8 +128,8 @@ export async function ensureRuntime(workdir: string, ctx: CodeGraphContext): Pro
128
128
  return { entry: runtimeCache, status };
129
129
  }
130
130
 
131
- export function registerCodeGraphTool(pi: ExtensionAPI): void {
132
- pi.registerTool({
131
+ export function registerCodeGraphTool(ext: ExtensionAPI): void {
132
+ ext.registerTool({
133
133
  name: "code_graph",
134
134
  label: "Code Graph",
135
135
  description: