pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
@@ -7,14 +7,11 @@ import * as os from "node:os";
7
7
  import * as path from "node:path";
8
8
  import { after, before, describe, it } from "node:test";
9
9
  import {
10
- applyCompleted,
11
10
  applyExecutionApproved,
12
11
  applyExecutionCompleted,
13
12
  applyExecutionProgress,
14
13
  applyExecutionHeadChanged,
15
14
  applyExecutionStopped,
16
- applyImplementationReviewConfigured,
17
- applyImplementationRoundFinished,
18
15
  applyLaneResult,
19
16
  applyMigration,
20
17
  applyPlanWritten,
@@ -331,33 +328,21 @@ describe("state machine reducers", () => {
331
328
  assert.throws(() => applyLaneResult(staged, "r2", "l1", { ok: true }), StateError);
332
329
  });
333
330
 
334
- it("implementation review: condition first, rounds counted, completed needs evidence (F-004)", () => {
335
- const { workdir, runId } = setupRun("sm-impl");
331
+ it("implementation-review write side is gone; legacy checkpoints stay readable (D-018)", () => {
332
+ const { workdir, runId } = setupRun("sm-impl-legacy");
336
333
  createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
337
334
  let cp = baseCheckpoint(workdir, runId);
338
- const planPath = path.join(path.dirname(checkpointFilePath(workdir, runId)!), "PLAN_v1.md");
339
- fs.writeFileSync(planPath, "# plan", "utf8");
340
- const plan = planIdentityOf(planPath, 1);
341
- cp = { ...cp, plan, phase: "implementation-review", nextAction: "ask-question" };
342
- assert.throws(
343
- () => applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] }),
344
- StateError,
345
- );
346
- cp = applyImplementationReviewConfigured(cp, "until-no-high");
347
- assert.throws(() => applyImplementationReviewConfigured(cp, "again"), StateError);
348
- cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 1, lanes: [{ laneId: "l1" }] });
349
- const file = writeReviewOutput(workdir, runId, "i1", "l1", "out");
350
- cp = applyLaneResult(cp, "i1", "l1", { ok: true, resultFile: file });
351
- assert.throws(() => applyImplementationRoundFinished(cp), StateError);
352
- cp = applyReviewConsolidated(cp, "i1");
353
- cp = applyImplementationRoundFinished(cp);
335
+ // Simulate a persisted 0.6.0 checkpoint in the legacy phase with its
336
+ // review bookkeeping — the schema still parses it read-only.
337
+ cp = {
338
+ ...cp,
339
+ phase: "implementation-review",
340
+ nextAction: "run-review",
341
+ implementationReview: { terminationCondition: "until-no-high", reviewerCount: 2, completedRounds: 1 },
342
+ };
343
+ const round = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 2, lanes: [{ laneId: "l1" }] });
344
+ assert.equal(round.reviewRounds.length, 1);
354
345
  assert.equal(cp.implementationReview?.completedRounds, 1);
355
- // Completion requires evidence AND a termination condition.
356
- assert.throws(() => applyCompleted({ ...cp, implementationReview: undefined }, "evidence"), StateError);
357
- assert.throws(() => applyCompleted(cp, " "), StateError);
358
- const done = applyCompleted(cp, "no findings in final round");
359
- assert.equal(done.phase, "completed");
360
- assert.equal(done.nextAction, "none");
361
346
  });
362
347
 
363
348
  it("execution approval and progress (D-003/D-011)", () => {
@@ -394,10 +379,12 @@ describe("state machine reducers", () => {
394
379
  assert.equal(cp.execution?.pausedReason, "stopped by user");
395
380
  const resumed = applyExecutionProgress(cp, { pausedReason: null });
396
381
  assert.equal(resumed.execution?.pausedReason, undefined);
397
- // Completion hands over to the implementation-review phase (R-005).
382
+ // Completion: v0.6.1 passes a completed audit straight to the terminal
383
+ // phase (the implementation-review loop is gone, D-018).
398
384
  const finished = applyExecutionCompleted(cp);
399
- assert.equal(finished.phase, "implementation-review");
400
- assert.equal(finished.nextAction, "ask-question");
385
+ assert.equal(finished.phase, "completed");
386
+ assert.equal(finished.nextAction, "none");
387
+ assert.equal(finished.execution?.audit?.passed, true);
401
388
  });
402
389
 
403
390
  it("migration resets rounds and approval, keeps termination (F-003)", () => {
@@ -431,72 +418,3 @@ describe("state machine reducers", () => {
431
418
  });
432
419
  });
433
420
 
434
- describe("implementationReview.reviewerCount (0.5.4)", () => {
435
- it("configured persists reviewerCount and survives checkpoint roundtrip", () => {
436
- const { workdir, runId } = setupRun("rc-persist");
437
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
438
- let cp = baseCheckpoint(workdir, runId);
439
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
440
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 2);
441
- assert.equal(cp.implementationReview?.reviewerCount, 2);
442
- assert.equal(cp.nextAction, "run-review");
443
- mutateCheckpoint(workdir, runId, () => cp);
444
- const reloaded = loadCheckpoint(workdir, runId);
445
- assert.ok(reloaded.status === "ok");
446
- assert.equal(reloaded.checkpoint.implementationReview?.reviewerCount, 2);
447
- });
448
-
449
- it("omitted reviewerCount stays undefined (legacy checkpoints unchanged)", () => {
450
- const { workdir, runId } = setupRun("rc-legacy");
451
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
452
- let cp = baseCheckpoint(workdir, runId);
453
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
454
- cp = applyImplementationReviewConfigured(cp, "1 round");
455
- assert.equal(cp.implementationReview?.reviewerCount, undefined);
456
- mutateCheckpoint(workdir, runId, () => cp);
457
- assert.ok(loadCheckpoint(workdir, runId).status === "ok");
458
- });
459
-
460
- it("rejects out-of-range and non-integer reviewerCount on load", () => {
461
- const { workdir, runId } = setupRun("rc-invalid");
462
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
463
- const file = checkpointFilePath(workdir, runId)!;
464
- const base = JSON.parse(fs.readFileSync(file, "utf8")) as { implementationReview?: unknown };
465
- for (const bad of [0, 4, "3", 1.5]) {
466
- const doc = {
467
- ...base,
468
- implementationReview: {
469
- terminationCondition: "1 round",
470
- reviewerCount: bad,
471
- completedRounds: 0,
472
- },
473
- };
474
- fs.writeFileSync(file, JSON.stringify(doc), "utf8");
475
- const loaded = loadCheckpoint(workdir, runId);
476
- assert.ok(loaded.status === "corrupt", `reviewerCount ${JSON.stringify(bad)} rejected`);
477
- }
478
- // Corrupt bytes refuse overwrite (mutateCheckpoint guard), which is the
479
- // intended fail-loud behavior — no restore attempted here.
480
- });
481
-
482
- it("applyMigration preserves an explicit reviewerCount (CQ1/D-4)", () => {
483
- const { workdir, runId } = setupRun("rc-migrate");
484
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
485
- let cp = baseCheckpoint(workdir, runId);
486
- cp = { ...cp, phase: "implementation-review", nextAction: "run-review" };
487
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
488
- cp = applyReviewRoundStarted(cp, { roundId: "i1", role: "reviewer", target: "implementation", reviewers: 3, lanes: [{ laneId: "l1" }] });
489
- const migrated = applyMigration(cp, { workdir: "/target/wt", worktreeRoot: "/target/wt", commonDir: "/target/.git" });
490
- assert.equal(migrated.implementationReview?.reviewerCount, 3);
491
- assert.equal(migrated.implementationReview?.completedRounds, 0);
492
- });
493
-
494
- it("second configuration write is rejected even with identical values (replay guard)", () => {
495
- const { workdir, runId } = setupRun("rc-replay");
496
- createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
497
- let cp = baseCheckpoint(workdir, runId);
498
- cp = { ...cp, phase: "implementation-review", nextAction: "ask-question" };
499
- cp = applyImplementationReviewConfigured(cp, "until-no-high", 3);
500
- assert.throws(() => applyImplementationReviewConfigured(cp, "until-no-high", 3), StateError);
501
- });
502
- });
@@ -1,9 +1,12 @@
1
1
  /**
2
2
  * `analyze_refs` tool — plan-with-refs per-reference analysis via read-only Pi
3
3
  * subagents with isolated context. One lane per reference (cwd = the ref's own
4
- * directory), reusing the reviewer role gates from `.git/pi_plans/config.json`
5
- * and the concurrent refinement overlay (title "Refs"). Batches are capped at
6
- * three concurrent lanes; larger ref sets run as sequential batches.
4
+ * directory), reusing the reviewer model confirmation from the GLOBAL config
5
+ * (`~/.pi/pi-plans/config.json`) and the concurrent overlay (title "Refs").
6
+ * analyze_refs is spawn-only by nature, so the reviewer MODE is deliberately
7
+ * not consulted here (Q-4=B): a current-session reviewer still gets spawned
8
+ * ref-analyst lanes, with a one-time notice in the result. Batches are capped
9
+ * at three concurrent lanes; larger ref sets run as sequential batches.
7
10
  *
8
11
  * Recording is best-effort: spawns land in `subagents.jsonl` (role
9
12
  * `ref-analyst`) only when an active planning run exists. Analysis output is
@@ -17,7 +20,18 @@ import { Text } from "@earendil-works/pi-tui";
17
20
  import { Type } from "typebox";
18
21
  import * as fs from "node:fs";
19
22
  import * as path from "node:path";
20
- import { loadConfig, normalizeWorkdir, readActive, recordSubagent, resolveStateRootOrNull, StateError } from "../src/state.ts";
23
+ import {
24
+ loadConfig,
25
+ normalizeWorkdir,
26
+ readActive,
27
+ recordSubagent,
28
+ resolveEffectiveReviewer,
29
+ resolveGlobalConfigPath,
30
+ resolveStateRootOrNull,
31
+ StateError,
32
+ } from "../src/state.ts";
33
+ import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type RolePanelHost } from "../src/role-panels.ts";
34
+ import { roleModelLabel } from "../src/thinking-levels.ts";
21
35
  import type { SubagentUsage } from "../src/subagent.ts";
22
36
  import { resolveActiveRun } from "../src/run-context.ts";
23
37
  import { buildRefAnalystTask, type RefAnalystTaskInput } from "../src/refine-prompts.ts";
@@ -45,25 +59,39 @@ const AnalyzeRefsParams = Type.Object({
45
59
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
46
60
  });
47
61
 
48
- function gateError(problem: "state" | "mode" | "current-session" | "confirm"): StateError {
62
+ function gateError(problem: "state" | "confirm", guidance?: string): StateError {
49
63
  if (problem === "state") {
50
64
  return new StateError("no pi-plans state found; run the plans tool (action: init) first");
51
65
  }
52
- if (problem === "mode") {
53
- return new StateError(
54
- "The reviewer role mode is missing or invalid in .git/pi_plans/config.json (analyze_refs reuses the reviewer gates). Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer).",
55
- );
56
- }
57
- if (problem === "current-session") {
58
- return new StateError(
59
- "The reviewer role mode is current-session, but analyze_refs only spawns delegated read-only subagents (one per reference). Ask the user to switch the reviewer mode to delegated-subagent via ask_choice, persist with the plans tool (set-role, role=reviewer, mode=delegated-subagent), then retry analyze_refs.",
60
- );
61
- }
62
66
  return new StateError(
63
- "The reviewer model was never confirmed (confirmed_at is null); analyze_refs reuses the reviewer confirmation. Ask the model-confirmation question with ask_choice: 1. Inherit the main agent's model (recommended) 2. Choose a model (list options from the /model picker; persist the exact provider/model selector) 3. Other 4. Auto-complete — then persist with the plans tool (set-role, role=reviewer, confirmed: true, modelSelector: the selector or 'inherit').",
67
+ guidance ??
68
+ firstUseTextGuidance([], resolveGlobalConfigPath()),
64
69
  );
65
70
  }
66
71
 
72
+ /** First-use model confirmation for the spawn-only ref-analyst path: native
73
+ * panels in TUI, menus for hasUI non-TUI, embedded text guidance otherwise.
74
+ * The reviewer MODE is not consulted (Q-4=B), but because analysis always
75
+ * spawns, a confirmed CONCRETE model is required even when the stored mode
76
+ * is current-session (model confirmation “as usual”). */
77
+ async function ensureRefAnalystModelReady(
78
+ host: RolePanelHost,
79
+ role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
80
+ ): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
81
+ if (role.confirmed_at !== null && role.model_selector !== null) return role as never;
82
+ let outcome = await runFirstUseFlow(host, role.thinking_level);
83
+ if (outcome.status === "confirmed" && outcome.model_selector !== null && availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
84
+ // F-008: a manually entered selector that the registry does not know —
85
+ // one re-pick, then let spawn-side errors surface precisely.
86
+ outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
87
+ }
88
+ if (outcome.status === "cancelled") throw firstUseCancelledError("analyze_refs");
89
+ const guidance = firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath());
90
+ if (outcome.status === "unavailable") throw gateError("confirm", guidance);
91
+ if (outcome.role.model_selector === null) throw gateError("confirm", guidance);
92
+ return outcome.role;
93
+ }
94
+
67
95
  interface AnalysisJob {
68
96
  input: RefAnalystTaskInput;
69
97
  name: string;
@@ -72,14 +100,14 @@ interface AnalysisJob {
72
100
  missing: string | null;
73
101
  }
74
102
 
75
- export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void {
103
+ export function registerAnalyzeRefsTool(ext: ExtensionAPI, baseDir: string): void {
76
104
  const agentPrompt = stripFrontmatter(fs.readFileSync(path.join(baseDir, "agents", "ref-analyst.md"), "utf8"));
77
105
 
78
- pi.registerTool({
106
+ ext.registerTool({
79
107
  name: "analyze_refs",
80
108
  label: "Analyze Refs",
81
109
  description:
82
- "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer role gates and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. Refuses until the reviewer mode/model gates pass in .git/pi_plans/config.json.",
110
+ "plan-with-refs: analyze downloaded references via independent read-only Pi subagents — one lane per reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config and the concurrent overlay. Batches of at most 3 lanes run sequentially; results are structured per-reference sections for REF_ANALYSIS.md. Recording into subagents.jsonl is best-effort (active run only); refs.jsonl stays owned by the main agent via the plans record-ref action. The reviewer mode is not consulted (spawn-only); first use pops native model/effort panels in TUI.",
83
111
  promptSnippet: "Analyze plan-with-refs references with per-ref read-only subagents",
84
112
  parameters: AnalyzeRefsParams,
85
113
 
@@ -92,17 +120,22 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
92
120
  throw gateError("state");
93
121
  }
94
122
  const config = loadConfig(root);
95
- const reviewer = config.reviewer;
96
- if (!reviewer || (reviewer.mode !== "delegated-subagent" && reviewer.mode !== "current-session")) {
97
- throw gateError("mode");
98
- }
99
- if (reviewer.mode === "current-session") {
100
- throw gateError("current-session");
101
- }
102
- if (reviewer.confirmed_at === null) {
103
- throw gateError("confirm");
123
+
124
+ // F-005: cheap validations BEFORE any first-use panel.
125
+ if (params.refs.length === 0) {
126
+ throw new StateError("analyze_refs requires at least one reference");
104
127
  }
105
128
 
129
+ // Effective reviewer from the global config (mode NOT consulted —
130
+ // analyze_refs is spawn-only, Q-4=B; a notice surfaces when the stored
131
+ // mode is current-session so the switch is never silent).
132
+ const { reviewer: initialReviewer } = resolveEffectiveReviewer(root);
133
+ const modeIgnoredNotice =
134
+ initialReviewer.mode === "current-session"
135
+ ? `note: the reviewer mode is ${initialReviewer.mode}, but analyze_refs always spawns read-only subagents; the mode is ignored here and unchanged.`
136
+ : null;
137
+ const reviewer = await ensureRefAnalystModelReady(ctx as unknown as RolePanelHost, initialReviewer);
138
+
106
139
  // Resolve refs and validate directories up front; missing ones become
107
140
  // FAILED sections instead of aborting the whole batch.
108
141
  const active = resolveActiveRun(ctx.sessionManager, workdir);
@@ -122,7 +155,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
122
155
  }
123
156
 
124
157
  const model = reviewer.model_selector ?? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined);
125
- const modelLabel = model ?? "inherit";
158
+ const modelLabel = roleModelLabel(model ?? "inherit", reviewer.thinking_level);
126
159
  let languageTag: string | null = null;
127
160
  if (active) {
128
161
  try {
@@ -140,6 +173,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
140
173
  role: "ref-analyst",
141
174
  name,
142
175
  model: okModel ?? model ?? null,
176
+ thinking_level: reviewer.thinking_level,
143
177
  // I-010: meter subagent token/cost for benchmark accounting.
144
178
  usage: usage
145
179
  ? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
@@ -160,6 +194,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
160
194
  task: buildRefAnalystTask({ ...job.input, languageTag }),
161
195
  cwd: job.dir,
162
196
  model,
197
+ thinkingLevel: reviewer.thinking_level ?? undefined,
163
198
  tools: READ_ONLY_TOOLS,
164
199
  signal: relay.signal,
165
200
  onProgress: (event) => overlay?.update(job.laneId, event),
@@ -224,7 +259,7 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
224
259
 
225
260
  if (failures === jobs.length) {
226
261
  throw new Error(
227
- `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) and re-ask the model-confirmation question.` : ""}`,
262
+ `all reference analysis subagents failed (${failures}/${jobs.length})${model ? `\nIf the model selector "${model}" is unavailable, reset the reviewer confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next analyze_refs opens the native model panel to re-confirm.` : ""}`,
228
263
  );
229
264
  }
230
265
 
@@ -237,13 +272,13 @@ export function registerAnalyzeRefsTool(pi: ExtensionAPI, baseDir: string): void
237
272
  content: [
238
273
  {
239
274
  type: "text",
240
- text: `${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
275
+ text: `${modeIgnoredNotice ? `${modeIgnoredNotice}\n\n` : ""}${text}\n\n---\nPersist: paste each reference's analysis into REF_ANALYSIS.md, call the plans tool (record-ref) per reference with coverage and gaps filled from the analysis, then ask at least three ref-specific adoption questions per reference with ask_choice before using its ideas in PLAN_v1.md.`,
241
276
  },
242
277
  ],
243
278
  details: {
244
279
  mode: "delegated-subagent",
245
280
  role: "ref-analyst",
246
- reviewerGates: { mode: reviewer.mode, model },
281
+ reviewerGates: { mode: initialReviewer.mode, modeIgnored: modeIgnoredNotice !== null, model, thinkingLevel: reviewer.thinking_level },
247
282
  batches: Math.ceil(jobs.length / BATCH_SIZE),
248
283
  model,
249
284
  outputs,
@@ -16,13 +16,6 @@ import { Text } from "@earendil-works/pi-tui";
16
16
  import { Type } from "typebox";
17
17
  import { disableAutoComplete, enableAutoComplete, isAutoCompleteEnabled, recordAskChoice } from "../src/autocomplete.ts";
18
18
  import { assertAutoApprovable, isAutoApproveEnabled } from "../src/auto-approve.ts";
19
- import {
20
- TERMINATION_QUESTION,
21
- TERMINATION_OPTIONS,
22
- TERMINATION_RECORDING_INSTRUCTIONS,
23
- implReviewerCountPromptLine,
24
- renderTerminationOptions,
25
- } from "../src/termination-prompt.ts";
26
19
  import { truncateToWidth, visibleWidth } from "../src/refine-ui-helpers.ts";
27
20
  import { stripRecommendedMarker } from "../src/ask-form.ts";
28
21
  import {
@@ -63,8 +56,8 @@ export const FALLBACK_ROWS = 30;
63
56
  /** Minimal-form floor for tiny terminals (stage-3 width). */
64
57
  const MINIMAL_LINE_WIDTH = 20;
65
58
  /**
66
- * Truncation floor for fixed tail labels (Other…/Auto-complete/Auto-refine
67
- * loop): the longest magic prefix ("Auto-refine loop", 16 cols) plus slack.
59
+ * Truncation floor for fixed tail labels (Other…/Auto-complete): the
60
+ * longest magic prefix ("Auto-complete", 14 cols) plus slack.
68
61
  * These labels drive startsWith() answer routing and must never lose it.
69
62
  */
70
63
  const FIXED_LABEL_FLOOR = 18;
@@ -74,7 +67,7 @@ export interface PanelItem {
74
67
  core: string;
75
68
  /** Full display label: core + description (degradation stage 0). */
76
69
  display: string;
77
- /** Fixed tail labels (Other…/Auto-complete/Auto-refine loop): truncation keeps at least the magic prefix. */
70
+ /** Fixed tail labels (Other…/Auto-complete): truncation keeps at least the magic prefix. */
78
71
  fixed?: boolean;
79
72
  }
80
73
 
@@ -156,7 +149,12 @@ export function fitAskChoicePanel(question: string, items: PanelItem[], columns:
156
149
  export const Option = Type.Object(
157
150
  {
158
151
  label: Type.String({ description: "Option label" }),
159
- description: Type.Optional(Type.String({ description: "Short tradeoff that matters, shown to the user" })),
152
+ description: Type.Optional(
153
+ Type.String({
154
+ description:
155
+ "REQUIRED on every option you author: '✓ <advantage> / ✗ <drawback>' — the user compares options side by side, so each one must state what it gains AND what it costs. Write BOTH halves in this single description string, in the configured language, tersely (≈8 words per half). If a side is genuinely absent write '—' rather than dropping it. Do NOT invent separate pros/cons fields: Option accepts no other keys.",
156
+ }),
157
+ ),
160
158
  recommended: Type.Optional(Type.Boolean({ description: "Mark exactly one recommended option; put it first. Never embed (推荐)/(recommended) text in labels — the UI renders the ★ marker automatically" })),
161
159
  },
162
160
  { additionalProperties: false },
@@ -165,7 +163,10 @@ export const Option = Type.Object(
165
163
  export const BatchQuestionParams = Type.Object(
166
164
  {
167
165
  question: Type.String({ description: "The question to ask, in the configured language" }),
168
- options: Type.Array(Option, { description: "Ordered options: recommended first, alternatives next. Do not include Other or Auto-complete yourself." }),
166
+ options: Type.Array(Option, {
167
+ description:
168
+ "Ordered options: recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself.",
169
+ }),
169
170
  allowOther: Type.Optional(Type.Boolean({ description: "Offer free-form input for this question (default true)" })),
170
171
  autoComplete: Type.Optional(
171
172
  Type.Boolean({
@@ -182,7 +183,7 @@ export const BatchQuestionParams = Type.Object(
182
183
  export const AskChoiceParams = Type.Object(
183
184
  {
184
185
  question: Type.Optional(Type.String({ description: "The single question to ask, in the configured language (mutually exclusive with questions)." })),
185
- options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Do not include Other or Auto-complete yourself." })),
186
+ options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself." })),
186
187
  questions: Type.Optional(
187
188
  Type.Array(BatchQuestionParams, {
188
189
  description:
@@ -205,12 +206,6 @@ export const AskChoiceParams = Type.Object(
205
206
  purpose: Type.Optional(
206
207
  Type.String({ description: "Short machine-readable purpose (e.g. 'scope', 'termination-condition')." }),
207
208
  ),
208
- trailing: Type.Optional(
209
- StringEnum(["auto-refine-loop"] as const, {
210
- description:
211
- 'Replace the trailing Auto-complete option with "Auto-refine loop" (post-execution amelioration prompt). Selecting it returns instructions to ask the rounds/termination follow-up; Auto-complete is suppressed entirely for this question.',
212
- }),
213
- ),
214
209
  workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
215
210
  },
216
211
  { additionalProperties: false },
@@ -581,16 +576,17 @@ function formatBatchAnswers(batch: NonNullable<AskChoiceDetails["batch"]>): stri
581
576
  }
582
577
  const NL = "\n";
583
578
 
584
- export function registerAskChoiceTool(pi: ExtensionAPI): void {
585
- pi.registerTool({
579
+ export function registerAskChoiceTool(ext: ExtensionAPI): void {
580
+ ext.registerTool({
586
581
  name: "ask_choice",
587
582
  label: "Ask Choice",
588
583
  description:
589
- "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt.",
584
+ "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the questionIds reserved for handoff. EVERY option you author — including the accept/execute handoff — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
590
585
  promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
591
586
  promptGuidelines: [
592
587
  "Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
593
588
  "Batch a round's questions into one ask_choice call (questions: [...], 2-8 items) instead of asking one at a time, then think after the answers and follow up with later calls. Scope confirmation and execution handoff are always separate single-question calls (autoComplete: false).",
589
+ "Give every option you author a description of the form '✓ <advantage> / ✗ <drawback>' — the user's whole point is seeing what each option wins and what it costs, in the configured language. Keep each half terse (~8 words). Put both halves in the description string; there are no separate pros/cons fields, and Other/Auto-complete are added by the tool.",
594
590
  ],
595
591
  parameters: AskChoiceParams,
596
592
  executionMode: "sequential",
@@ -604,9 +600,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
604
600
  if (params.question !== undefined || params.options !== undefined) {
605
601
  throw new Error("ask_choice accepts either question+options or questions, not both");
606
602
  }
607
- if (params.trailing !== undefined) {
608
- throw new Error("ask_choice batch mode does not support trailing (single-question only)");
609
- }
610
603
  return executeAskChoiceBatch({ questions: params.questions, workdir: params.workdir }, ctx);
611
604
  }
612
605
  const workdir = normalizeWorkdir(params.workdir ?? ctx.cwd);
@@ -645,10 +638,7 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
645
638
  /* the decisions ledger already holds the answer; F-005 reconcile covers the gap */
646
639
  }
647
640
  };
648
- // Param normalization: a trailing option replaces Auto-complete entirely,
649
- // so an erroneously passed autoComplete flag is suppressed here.
650
- const trailing = params.trailing;
651
- const autoComplete = (params.autoComplete ?? true) && trailing === undefined;
641
+ const autoComplete = params.autoComplete ?? true;
652
642
  const options = params.options;
653
643
  if (options.length === 0) throw new Error("ask_choice requires at least one option");
654
644
  const recommended = options.find((option) => option.recommended) ?? options[0];
@@ -748,8 +738,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
748
738
  };
749
739
  }
750
740
 
751
- const AUTO_REFINE_LOOP_LABEL =
752
- "Auto-refine loop (run refinement rounds until no high-severity finding or the 5-round cap)";
753
741
  const panelItems: PanelItem[] = options.map((option, index) => {
754
742
  const label = stripRecommendedMarker(option.label);
755
743
  const isRec = option === recommended;
@@ -761,7 +749,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
761
749
  });
762
750
  if (allowOther) panelItems.push({ core: "Other… (type your own answer)", display: "Other… (type your own answer)", fixed: true });
763
751
  if (autoComplete) panelItems.push({ core: "Auto-complete (take the recommended option)", display: "Auto-complete (take the recommended option)", fixed: true });
764
- else if (trailing) panelItems.push({ core: AUTO_REFINE_LOOP_LABEL, display: AUTO_REFINE_LOOP_LABEL, fixed: true });
765
752
 
766
753
  const panel = fitAskChoicePanel(
767
754
  params.question,
@@ -804,23 +791,6 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
804
791
  };
805
792
  }
806
793
 
807
- if (trailing && selected.startsWith("Auto-refine loop")) {
808
- recordAskChoice(ctx, false);
809
- record("Auto-refine loop", "user");
810
- // Skill-aware reviewer-count default (D-1/D-4): same mapping the
811
- // goal-running continuation in src/exec.ts renders.
812
- const activeSkill = resolveActiveRun(ctx.sessionManager, workdir)?.skill;
813
- return {
814
- content: [
815
- {
816
- type: "text",
817
- text: `User selected Auto-refine loop. Immediately ask the follow-up with ask_choice (autoComplete: false, in the session language): "${TERMINATION_QUESTION}" Options (recommended first): ${renderTerminationOptions()}. ${TERMINATION_RECORDING_INSTRUCTIONS} ${implReviewerCountPromptLine(activeSkill)} Then run the loop per the completion instructions: each round calls refine (role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>), accepts findings on evidence, applies fixes, re-runs relevant tests, and continues until the chosen termination condition — the goal-wait option keeps the loop running until no unpassed VCs remain.`,
818
- },
819
- ],
820
- details: details("Auto-refine loop", "user"),
821
- };
822
- }
823
-
824
794
  if (allowOther && selected.startsWith("Other…")) {
825
795
  const typed = await ctx.ui.input(`${params.question} — your answer:`);
826
796
  if (typed === undefined || !typed.trim()) {
@@ -128,8 +128,8 @@ export async function ensureRuntime(workdir: string, ctx: CodeGraphContext): Pro
128
128
  return { entry: runtimeCache, status };
129
129
  }
130
130
 
131
- export function registerCodeGraphTool(pi: ExtensionAPI): void {
132
- pi.registerTool({
131
+ export function registerCodeGraphTool(ext: ExtensionAPI): void {
132
+ ext.registerTool({
133
133
  name: "code_graph",
134
134
  label: "Code Graph",
135
135
  description: