pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
package/AGENTS.md ADDED
@@ -0,0 +1,58 @@
1
+ # AGENTS.md
2
+
3
+ Authoritative conventions for AI agents and human contributors working in this
4
+ repository. Where another document disagrees with this one on **documentation
5
+ language** or **commit messages**, this file wins.
6
+
7
+ ## Documentation language
8
+
9
+ **All prose documentation in this repository is written in English.** This
10
+ applies to every Markdown file that ships in the package — `CHANGELOG.md`,
11
+ `README.md`, `CONTRIBUTING.md`, this file, everything under `references/`,
12
+ `skills/`, and `agents/`.
13
+
14
+ - `CHANGELOG.md` is **entirely English**, including text inside inline code
15
+ spans. When an entry documents a Chinese-locale UI string, restate it
16
+ semantically in English rather than quoting the Chinese literal.
17
+ - Identifiers stay verbatim regardless of the surrounding language: file
18
+ paths, function and type names, CLI flags, environment variables, error
19
+ messages, config keys, run ids, and version numbers are copied byte-for-byte
20
+ and never translated.
21
+ - `scripts/validate.ts` enforces "no CJK ideographs and no full-width CJK
22
+ punctuation in `CHANGELOG.md`". If that check ever blocks a legitimate
23
+ entry, the rule is wrong — fix the rule deliberately rather than widening
24
+ the regex ad hoc.
25
+
26
+ ### Not documentation
27
+
28
+ The localized UI strings in `src/ui-language.ts` and the CJK fixtures in
29
+ `tests/` are **features, not prose**. Chinese is a supported UI locale and the
30
+ i18n tables must keep their `zh` entries; test fixtures assert on real Chinese
31
+ strings. Do not "translate" either of them.
32
+
33
+ ## Commit messages
34
+
35
+ **Both the subject and the body must be English.**
36
+
37
+ Follow the existing style: `type: short imperative description` (a scope is
38
+ optional, e.g. `fix(form): ...`).
39
+
40
+ - `feat:` new feature
41
+ - `fix:` bug fix
42
+ - `docs:` documentation
43
+ - `refactor:` no behavior change
44
+ - `perf:` performance
45
+ - `test:` tests only
46
+ - `chore:` housekeeping
47
+ - `ci:` CI changes
48
+
49
+ Use the imperative mood ("fix race in ...", not "fixed ..."), keep the subject
50
+ short, and use the body for motivation and evidence. Do not add generator or
51
+ `Co-Authored-By` trailers.
52
+
53
+ ## Edit authority
54
+
55
+ This file governs **language and commit-message conventions only**. It grants no
56
+ permission to modify any file. Who may edit what — in particular the
57
+ `CHANGELOG.md` rule — is defined in `CONTRIBUTING.md`, which remains
58
+ authoritative on that point.
package/CONTRIBUTING.md CHANGED
@@ -28,15 +28,15 @@ pi-plans/
28
28
  │ └── code-graph/ # SQLite schema/store, parsers, indexer, summary, materialize
29
29
  ├── skills/ # Planning router plus five specialist planning skills
30
30
  ├── references/ # Shared workflow, state/config, plan template (normative)
31
- ├── agents/ # reviewer.md / criticizer.md subagent prompts (read-only)
31
+ ├── agents/ # reviewer.md subagent prompt (read-only; ref-analyst.md for reference analysis)
32
32
  ├── scripts/ # validate.ts (structure + package artifact guard), run-tests.ts
33
33
  └── tests/ # node:test suite
34
34
  ```
35
35
 
36
36
  `npm run validate` enforces several invariants, so keep them intact:
37
37
 
38
- - every directory under `skills/` has a `SKILL.md` with frontmatter (`name` matching the directory, routing language in `description`) and the required phrases (`ask_choice`, `refine`, `.git/pi_plans`, ...)
39
- - the `reviewer` and `criticizer` agent prompts declare read-only tools and state the read-only contract
38
+ - every directory under `skills/` has a `SKILL.md` with frontmatter (`name` matching the directory, routing language in `description`) and the required phrases (`ask_choice`, `refine`, `.git/pi-plans`, ...)
39
+ - the `reviewer` agent prompt declares read-only tools and states the read-only contract
40
40
  - the npm artifact stays code-sized: no `scripts/bench/vendor|results` entries, unpacked < 5 MiB, packed < 3 MiB, and key entries present
41
41
  - `package.json` metadata (license, `pi-package` keyword, engines, scripts, required `files`) stays as asserted
42
42
 
@@ -71,18 +71,11 @@ Documentation duty: if your change alters behavior or the public API, update `RE
71
71
 
72
72
  ## Commit messages
73
73
 
74
- Follow the existing style: `type: short imperative description` (a scope is optional, e.g. `fix(form): ...`).
75
-
76
- - `feat:` new feature
77
- - `fix:` bug fix
78
- - `docs:` documentation
79
- - `refactor:` no behavior change
80
- - `perf:` performance
81
- - `test:` tests only
82
- - `chore:` housekeeping
83
- - `ci:` CI changes
84
-
85
- Use the imperative mood ("fix race in ...", not "fixed ..."), keep the subject short, and use the body for motivation and evidence. Do not add generator or `Co-Authored-By` trailers.
74
+ Commit-message format and the repository's documentation-language rule are
75
+ specified in [`AGENTS.md`](AGENTS.md), which is authoritative for both. In short:
76
+ subject **and** body are English, in the style `type: short imperative
77
+ description` (a scope is optional, e.g. `fix(form): ...`), with the body used
78
+ for motivation and evidence and no generator or `Co-Authored-By` trailers.
86
79
 
87
80
  ## Issues and pull requests
88
81
 
package/README.md CHANGED
@@ -22,7 +22,7 @@
22
22
 
23
23
  ---
24
24
 
25
- A rough change request becomes a versioned Markdown plan instead of a surprise diff. The agent inspects your repository read-only, asks scoped planning questions one at a time, and stores every answer in a per-run ledger. Reviewer and criticizer subagents refine the plan until it converges — and only after you explicitly approve the handoff does the extension enter a tracked execution loop that injects the remaining verifier checklist every turn and lifts the write guard. Nothing outside planning artifacts is writable until that approval.
25
+ A rough change request becomes a versioned Markdown plan instead of a surprise diff. The agent inspects your repository read-only, asks scoped planning questions one at a time, and stores every answer in a per-run ledger. Reviewer subagents refine the plan (findings plus questions) until it converges — and only after you explicitly approve the handoff does the extension enter a task-tree execution loop that injects the current wave and remaining tasks every turn, tracks progress through the `plans_update_task` tool, gates completion on an independent audit, and lifts the write guard. Nothing outside planning artifacts is writable until that approval.
26
26
 
27
27
  ## Benchmarked: 6× more tasks solved
28
28
 
@@ -62,29 +62,31 @@ at <b>7.6× fewer tokens per solved task</b>.
62
62
  language + docs location (once per workspace)
63
63
  |
64
64
  planning questions, one ask_choice at a time | write guard ON
65
- | | only .git/pi_plans/,
65
+ | | only .git/pi-plans/,
66
66
  v | run artifacts, cache
67
- PLAN_vN.md + Verifier Checklist | are writable
67
+ PLAN_vN.md (## Tasks + ## Verification
68
+ Checks) | are writable
68
69
  ^ |
69
70
  | refine rounds |
70
71
  +----------------+
71
- reviewer (x1..x3) -> criticizer -> revise
72
+ reviewer (x1..x3): findings + questions
72
73
  |
73
74
  v
74
75
  explicit approval (never auto-completed)
75
76
  |
76
77
  =============================================== write guard OFF
77
78
  |
78
- tracked execution loop
79
+ task-tree execution loop
79
80
  fused AGENTS.md × Ponytail executor rules
80
- checklist injected each turn, [DONE:VC-xxx]
81
- markers tracked via bottom status bar
81
+ current wave + tasks injected each turn,
82
+ plans_update_task reports status + evidence,
83
+ execution reviewer verifies every check
82
84
  |
83
85
  v
84
86
  run status: done
85
87
  ```
86
88
 
87
- Every plan version carries stable IDs (`I-###`, `VC-###`) that never get recycled across revisions, so acceptance criteria survive refinement rounds intact.
89
+ Every plan version carries stable IDs (`Task-N`, `VC-###`) that never get recycled across revisions, so the task tree and verification checks survive refinement rounds intact.
88
90
 
89
91
  ## Quick start
90
92
 
@@ -102,12 +104,12 @@ Then describe a change from any repository:
102
104
  You: Create a plan to split the execution loop into smaller modules.
103
105
 
104
106
  Pi: Which planning docs location should this workspace use?
105
- 1. ./docs/pi-plans (recommended)
106
- 2. ./.git/pi_plans/plans
107
+ 1. ./.git/pi-plans/plans (recommended)
108
+ 2. ./docs/pi-plans
107
109
  3. Other
108
110
  4. Auto-complete
109
111
 
110
- Pi: Wrote ./docs/pi-plans/2026-08-26-split-execution-loop/PLAN_v1.md
112
+ Pi: Wrote ./.git/pi-plans/plans/2026-08-26-split-execution-loop/PLAN_v1.md
111
113
  Example verifier item:
112
114
  - [ ] `VC-001` covers `I-001`; pass condition: `npm test` passes;
113
115
  evidence: test output; metric: zero failing tests.
@@ -121,26 +123,26 @@ Pi: Accept the plan and execute it now?
121
123
  You: 1 — accept and execute.
122
124
  ```
123
125
 
124
- Planning artifacts live under `./docs/pi-plans/YYYY-MM-DD-<topic>/` by default (public, committed). Prefer `.git/pi_plans/plans` if you want them private to the repository.
126
+ Planning artifacts live under `./.git/pi-plans/plans/YYYY-MM-DD-<topic>/` by default — private to the repository, never tracked, never published. Choose `./docs/pi-plans` instead if you want the plans public and committed alongside the code.
125
127
 
126
128
  ## What it does
127
129
 
128
130
  | Capability | In short |
129
131
  |---|---|
130
132
  | Planning router + five specialist skills | Start with `/skill:planning` to route to the narrowest matching specialist (`plan-small` → `plan-big`, `debug-and-plan`, `plan-with-refs`) |
131
- | Choice prompts | `ask_choice`: recommended option first, answers auto-recorded per run; every option you author states its advantage and its drawback as `✓ <advantage> / ✗ <drawback>` in the configured language, so the user can weigh each option before answering; choosing Auto-complete enables recommendation-only answers for later eligible questions in the current planning run, with `/plans-autocomplete-stop` available to take back control. After execution completes, the continuation prompt enters goal-running mode by default: it asks only for the implementation-review loop's termination condition and then keeps refining until the loop ends or the cap is reached. |
132
- | Refinement rounds | Read-only reviewer/criticizer Pi subagents consolidate findings into the next plan version; delegated runs have standalone `Reviewer`/`Criticizer` progress overlays that close before the tool result returns; `analyze_refs` shows the same kind of overlay titled `Refs` while per-reference analysis subagents run |
133
- | Workspace state | Config, runs, decisions, refs, and subagent ledgers in `.git/pi_plans/` (git common dir) |
133
+ | Choice prompts | `ask_choice`: recommended option first, answers auto-recorded per run; every option you author states its advantage and its drawback as `✓ <advantage> / ✗ <drawback>` in the configured language, so the user can weigh each option before answering; choosing Auto-complete enables recommendation-only answers for later eligible questions in the current planning run, with `/plans-autocomplete-stop` available to take back control. |
134
+ | Refinement rounds | Read-only reviewer Pi subagents return findings (`F-###`) and up to five questions (`Q-1..Q-5`) in one round; the main agent asks every question with `ask_choice` and records the answers before revising. Delegated runs have a standalone `Reviewer` progress overlay; `analyze_refs` shows the same kind of overlay titled `Refs` while per-reference analysis subagents run |
135
+ | Workspace state | Config, runs, decisions, refs, and subagent ledgers in `.git/pi-plans/` (git common dir) |
134
136
  | VCC compact | Active planning/execution compaction uses deterministic, no-LLM VCC-style summaries when Pi core emits manual `/compact`, threshold, or overflow events. Summaries use five bracket sections plus a brief transcript, keep a smart recent tail, support `keep:N`, and write VCC details/stats without adding `/pi-vcc` commands. |
135
- | Visible Refiner overlay | Delegated reviewer/criticizer subagents surface as a named public overlay in the TUI — one `Reviewer`/`Criticizer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
136
- | Tracked execution | Checklist injected each turn; `[DONE:VC-xxx]` markers drive completion; implementation items report progress with `[I-xxx:implemented]` / `[I-xxx:validating]` markers; the bottom status bar shows lifecycle, `x/y` progress, elapsed time, and input/output token usage in real time |
137
+ | Visible Refiner overlay | Delegated reviewer subagents surface as a named public overlay in the TUI — one `Reviewer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
138
+ | Tracked execution | The current wave and remaining tasks are injected each turn; task progress is reported exclusively through the `plans_update_task` tool (status + evidence / skipReason, audit-only rollback); the task dashboard shows the tree live (compact aboveEditor widget, Ctrl+Shift+T expanded view with ✓/▸/~/· markers, width-adaptive); a stall watchdog pauses after three settled rounds without task-state change; the status bar shows lifecycle, `x/y` task progress, elapsed time, and token usage in real time |
137
139
  | Multi-run workdirs (0.6.0) | Several pi sessions can plan concurrently in one workdir: the run registry derives from `runs/` (no shared pointer to race), each session binds to its run, and same-topic runs get suffixed artifact dirs. `/plans-abandon`, `/plans-execute`, and `/resume-plans` are binding-first and open a descriptive run-picker form when more than one candidate exists; `/plans` lists all runs (newest first, bound run marked) |
138
- | Goal-wait continuation | In TUI/RPC, only a fully settled agent with unpassed VCs and no pending input or compaction gets one hidden wake carrying the latest checklist. Tool turns never queue reminders or consume guard rounds. The status bar shows progress; 3 no-progress cycles or 6 waiting cycles pause continuation. User interruption and final model errors also pause it. Only genuine user input or `/plans-execute` resumes; extension messages cannot. Print/JSON single-shot sessions track progress without automatic wakes. `/plans-stop` terminates execution |
139
- | Execution handoff | The accepted plan resumes in the current session (recommended) or, after the runtime question, on a **delegated executor subagent** running a different model: ≥3 switch targets (session-visible models + registry, deduped), one child for the whole plan with native write tools, `Executor` overlay progress, parent-side `[DONE:VC-xxx]` tracking, Esc//`/plans-stop` → stopped (resumable), and a configurable timeout (`executor_timeout_minutes`, default 60). Auto-approve/no-UI skips the question (current session, `[auto-approve]` recorded) |
140
- | Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current `I-###`, implementation IDs, and remaining `VC-###` checklist into the VCC sections. The old current-I proactive trigger and model-generated summary path are removed. |
140
+ | Execution reviewer | When every task reaches a terminal state, the run status moves to `verifying` and an independent read-only reviewer verifies each `VC-###` check in a detached, overlay-visible round (Esc closes; Ctrl+Shift+R reopens the in-flight round); failed checks roll their covered tasks (children cascade, skipped reopen) back to pending and inject the review report; five committed rounds bound the loop — exhaustion pauses in every mode, and only an explicit `/plans-execute` confirmation grants a fresh budget (ordinary input and restores never refill). Checks with all-skipped coverage pass; checks covering no task never audit |
141
+ | Execution handoff | The accepted plan executes in the current session after explicit approval (never auto-completed); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; 0.6.0 in-flight runs resume compatibly (delegated-executor orphans re-approve, paused executions rebuild from the task tree) |
142
+ | Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current task, task ids, and remaining `VC-###` checks into the VCC sections. Proactive triggers and model-generated summary paths are removed. |
141
143
  | Planning-phase compaction | During `run.status=planning` with no active execution, pi-plans maps active run, artifact directory, latest plan path from session entries, and observed current-I markers into the VCC sections. Without an active planning run, compaction returns to Pi core. Additionally, creating a new run (`plans start-run`) proactively requests one pre-plan VCC compaction and resumes planning with a hidden message (default on; `prePlanCompact:false` disables). |
142
144
  | Efficient executor prompt | Each turn, the executor is steered by a fused rule set — Marcos Hernanz's AGENTS.md principles × Ponytail minimalism: layered growth, simplest implementation, long-term architecture (no stopgaps), library discipline — so plans finish in fewer tokens and fewer detours |
143
- | Write guard | `edit`/`write` blocked outside planning artifacts while a run is active; delegated executor children (and `PI_PLANS_RUN_ID`-pinned children) are exempt so a foreign session's planning run can never block an executing child; graph-aware file tools fall back to native writes for executor children |
145
+ | Write guard | `edit`/`write` blocked outside planning artifacts while a planning run is active in the workdir; the guard prefers the session-bound run and lists the allowed roots on refusal |
144
146
 
145
147
  ## Interface overview
146
148
 
@@ -148,16 +150,16 @@ Planning artifacts live under `./docs/pi-plans/YYYY-MM-DD-<topic>/` by default (
148
150
  |---|---|
149
151
  | `plans` | State CLI: `init`, `show`, `set-language`, `set-artifact-root`, `set-refs-root`, `set-role`, `start-run`, `set-status`, `record-decision`, `record-ref`, `record-subagent`, `record-checkpoint` (state-machine-validated workflow transitions) |
150
152
  | `ask_choice` | Numbered choice prompt; `autoComplete: false` for the merged accept/execute question and external-state questions |
151
- | `refine` | Reviewer/criticizer round via standalone read-only subagents (`--mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` for both roles when the workspace has the code graph enabled); `target: "plan"` (default) reviews the plan, `target: "implementation"` reviews the implemented worktree against the plan; delegated TUI runs show one `Reviewer`/`Criticizer` overlay (78% width × 78% height, top-center, ≥72 cols) with per-lane transcript, follow-bottom scroll, Tab focus, and retention until `Esc`; `reviewers: 3` renders three equal-height panes; enforces role/model confirmation gates |
152
- | `analyze_refs` | plan-with-refs reference analysis: one independent read-only subagent per downloaded reference (cwd = the ref directory), reusing the reviewer role gates and the concurrent overlay (titled `Refs`); batches of at most 3 lanes run sequentially; returns structured per-reference sections for `REF_ANALYSIS.md` |
153
- | `execute_plan` | Execution handoff: re-confirms with the user, asks the runtime (current session vs switch model → delegated executor), and enters extension-managed execution mode; picks the run via a descriptive form when several planned runs coexist |
153
+ | `refine` | Reviewer round via standalone read-only subagents (`--mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` when the workspace has the code graph enabled): findings (`F-###`) and up to five questions (`Q-1..Q-5`) per lane; the caller must ask every question with `ask_choice` and record answers before revising; delegated TUI runs show one `Reviewer` overlay (78% width × 78% height, top-center, ≥72 cols) with per-lane transcript, follow-bottom scroll, Tab focus, and retention until `Esc`; `reviewers: 3` renders three equal-height panes; enforces the reviewer gates — first use pops native model + effort panels in TUI (menus on RPC, text guidance headless), persisted to the global reviewer config |
154
+ | `analyze_refs` | plan-with-refs reference analysis: one independent read-only subagent per downloaded reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config (the mode is not consulted — analysis always spawns) and the concurrent overlay (titled `Refs`); batches of at most 3 lanes run sequentially; returns structured per-reference sections for `REF_ANALYSIS.md` |
155
+ | `execute_plan` | Execution handoff: re-confirms with the user (never auto-completed) and enters task-tree execution mode (`plans_update_task` progress, dashboard, execution reviewer); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; picks the run via a descriptive form when several planned runs coexist |
154
156
  | `/plans` | Show config, all runs (newest first, bound run marked, cap 50), and execution progress |
155
- | `/config-pi-plans` | Re-ask workspace defaults for language, artifact root, refs root, code graph, reviewer mode/model, and criticizer mode/model |
156
- | `/resume-plans` | Resume a run in the CURRENT session across restarts: unfinished planning (pending question + answered decisions), reviewing (round/lane state, successful outputs reused), execution (approval digest + HEAD + verified VCs), and implementation review (termination condition + round count). Binding-first: the session-bound resumable run resumes directly; a unique candidate goes direct; multiple candidates get a descriptive chooser. Linked worktrees share candidates; a cross-worktree resume confirms, copies artifacts without overwriting, and resets approval + VC validity (the termination condition survives, rounds restart at 0). An unchanged plan digest with a changed HEAD keeps the authorization but re-verifies old VCs first. Busy sessions and runs actively owned by a live process only notify — no queueing, no takeover. Interactive (TUI/RPC) only |
157
- | `/plans-execute [plan.md]` | Resume a paused active execution without losing verified progress; otherwise enter the explicit execution handoff (run-picker form when several planned runs coexist; runtime question: current session or switch model) |
157
+ | `/config-pi-plans` | Re-ask workspace defaults for language, artifact root, refs root, and code graph, plus the reviewer mode/model (keep/change menu; native model + effort panels on change in TUI; current-session skips the model step) |
158
+ | `/resume-plans` | Resume a run in the CURRENT session across restarts: unfinished planning (pending question + answered decisions), reviewing (round/lane state, successful outputs reused), and execution (approval digest + HEAD + recorded task progress; 0.6.0 delegated-executor orphans re-approve, legacy implementation-review phases map to done). Binding-first: the session-bound resumable run resumes directly; a unique candidate goes direct; multiple candidates get a descriptive chooser. Linked worktrees share candidates; a cross-worktree resume confirms, copies artifacts without overwriting, and resets approval + task progress. An unchanged plan digest with a changed HEAD keeps the authorization but re-opens closed tasks. Busy sessions and actively owned runs only notify — no queueing, no takeover. Interactive (TUI/RPC) only |
159
+ | `/plans-execute [plan.md]` | Resume a paused active execution without losing task progress; otherwise enter the explicit execution handoff (run-picker form when several planned runs coexist; legacy plans get an upgrade notice) |
158
160
  | `/update-plan [plan.md] [reason…]` | Interrupt-and-refine: stops execution (if any), returns the run to planning, and directs the agent to revise the plan into `PLAN_vN+1.md` while preserving verified work |
159
161
  | `/plans-autocomplete-stop` | Stop the current run's Auto-complete mode and return later planning questions to normal interaction |
160
- | `/init-graph` | Build the code graph: tree-sitter function index + cross-file call/import edges (EXTRACTED vs INFERRED confidence) + label-propagation communities; writes `.git/pi_plans/graph/GRAPH_REPORT.md` (subsystems, god nodes, edge stats). Full rebuilds never touch files with staged edits (fail-closed pending guard) |
162
+ | `/init-graph` | Build the code graph: tree-sitter function index + cross-file call/import edges (EXTRACTED vs INFERRED confidence) + label-propagation communities; writes `.git/pi-plans/graph/GRAPH_REPORT.md` (subsystems, god nodes, edge stats). Full rebuilds never touch files with staged edits (fail-closed pending guard) |
161
163
  | `/update-graph` | Incrementally reindex changed files (shared path used by apply/final-commit triggers and the watcher) |
162
164
  | `/apply-graph` | Materialize DB-first staged edits to the worktree; auto-reindexes the materialized set afterward |
163
165
  | `/graph-status` `/graph-drift` | Graph inventory and DB↔source convergence |
@@ -177,11 +179,11 @@ Pi core remains the owner of compaction scheduling: manual `/compact`, threshold
177
179
  - **Manual matrix.** Plain `/compact` and `/compact keep:N` compact and show stats without continuing. `/compact <text>` and `/compact keep:N <text>` compact, then send the text once as the follow-up prompt. Internal pi-plans compaction markers are never reused as user follow-up prompts.
178
180
  - **Fallbacks and stats.** Unsafe manual/threshold cuts cancel with a warning; overflow or retrying unsafe cuts return control to Pi core. Successful VCC compactions notify with kept-tail and summarized-message stats. Threshold/overflow compactions may queue one hidden continuation only when the running Pi version still needs it and `continueAfterThresholdCompact` is enabled.
179
181
  - **Pre-plan compaction.** When `plans start-run` creates a new planning run, pi-plans proactively requests one VCC compaction (internal hint `pi-plans planning pre-plan compact`) right after the run is created and before the first planning question, then resumes the planning turn with a hidden message — so each new plan starts on a lean context (LLM reasoning degrades with longer input). Small sessions, already-compacted sessions, and failures skip silently and still resume. `prePlanCompact:false` in the repo-private config restores the old behavior.
180
- - **Repo-private config.** Defaults are scaffolded in `.git/pi_plans/pi-vcc-config.json` under the resolved git common dir: `overrideDefaultCompaction:true`, `smartKeepTail:true`, `continueAfterThresholdCompact:true`, `prePlanCompact:true`, `debug:false`. Global pi-vcc config and `PI_VCC_CONFIG_PATH` are intentionally ignored.
182
+ - **Repo-private config.** Defaults are scaffolded in `.git/pi-plans/pi-vcc-config.json` under the resolved git common dir: `overrideDefaultCompaction:true`, `smartKeepTail:true`, `continueAfterThresholdCompact:true`, `prePlanCompact:true`, `debug:false`. Global pi-vcc config and `PI_VCC_CONFIG_PATH` are intentionally ignored.
181
183
 
182
184
  ## Visible Refiner overlay
183
185
 
184
- Delegated `refine` rounds (reviewer or criticizer) and `analyze_refs` rounds (titled `Refs`) show their progress directly inside the Pi TUI instead of disappearing into the child process's terminal. The overlay is a public, named panel so users always know who is doing what:
186
+ Delegated `refine` rounds (reviewer) and `analyze_refs` rounds (titled `Refs`) show their progress directly inside the Pi TUI instead of disappearing into the child process's terminal. The overlay is a public, named panel so users always know who is doing what:
185
187
 
186
188
  - **Pi-btw-aligned geometry.** Each round uses `width: "78%"`, `minWidth: 72`, `maxHeight: "78%"`, `anchor: "top-center"`, and `{ top: 1, left: 2, right: 2 }` margins (no dependency on `pi-btw`; the renderer is built on Pi's public `pi-tui` primitives).
187
189
  - **Complete streaming transcript.** Assistant text, thinking blocks, tool calls, tool results, and stderr are merged per turn/content block into lane entries without overlay-facing truncation; only the viewport slices them. Final `message_end` / `tool_execution_end` overwrite the live snapshot with the authoritative content.
@@ -216,8 +218,8 @@ Invoked via `resources_discover`, callable as `/skill:<name>`, directly as `/<na
216
218
  | Skill | Use it when |
217
219
  |---|---|
218
220
  | [`planning`](skills/planning/SKILL.md) | General router; selects the narrowest specialist skill before planning starts |
219
- | [`plan-small`](skills/plan-small/SKILL.md) | Small scoped change; 1–3 questions; one criticizer round |
220
- | [`plan-normal`](skills/plan-normal/SKILL.md) | Broad or risky change; 5–10 questions; reviewer + criticizer rounds |
221
+ | [`plan-small`](skills/plan-small/SKILL.md) | Small scoped change; 1–3 questions; one reviewer round |
222
+ | [`plan-normal`](skills/plan-normal/SKILL.md) | Broad or risky change; 5–10 questions; reviewer rounds |
221
223
  | [`plan-big`](skills/plan-big/SKILL.md) | Open-ended/high-risk effort; 10+ questions; three concurrent reviewers |
222
224
  | [`debug-and-plan`](skills/debug-and-plan/SKILL.md) | Bug, CI failure, regression, incident — diagnose before planning |
223
225
  | [`plan-with-refs`](skills/plan-with-refs/SKILL.md) | External references must be analyzed before planning — repos, papers (arXiv), engineering blogs, and docs sites all count; theoretical references are equal citizens. plan-normal/plan-big may optionally cite 1–2 search-found references without downloading |
@@ -277,16 +279,16 @@ pi-plans/
277
279
  │ └── code-graph/ # SQLite schema/store, parsers, indexer, summary, materialize
278
280
  ├── skills/ # The planning router plus five specialist planning skills
279
281
  ├── references/ # Shared workflow, state/config, plan template (normative)
280
- ├── agents/ # reviewer.md / criticizer.md subagent prompts
282
+ ├── agents/ # reviewer.md subagent prompt (ref-analyst.md for reference analysis)
281
283
  ├── scripts/validate.ts # Structure + package artifact guard
282
284
  └── tests/ # node:test suite (state, guard, plan parsing, execution, refine progress, code-graph)
283
285
  ```
284
286
 
285
287
  ## Safety model
286
288
 
287
- Before the approved handoff the workflow writes only `.git/pi_plans/` state, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root (set via `plans set-refs-root` or `/config-pi-plans`; the recommended `.git/pi-plans/refs/` lives inside the git dir and needs no extra guard) — the extension blocks `edit`/`write` elsewhere while a run is `planning`/`accepted` (bash stays discipline-bound: inspection, `git init`, downloads into the cache). Reviewer/criticizer/ref-analyst subagents run with read-only tools. `Auto-complete` may answer planning and refinement questions only; it is never offered for execution, installs, publishing, deployment, merge, push, or credential use, and non-interactive sessions stop instead of auto-approving those.
289
+ Before the approved handoff the workflow writes only `.git/pi-plans/` state, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root (set via `plans set-refs-root` or `/config-pi-plans`; the recommended `.git/pi-plans/refs/` lives inside the git dir and needs no extra guard) — the extension blocks `edit`/`write` elsewhere while a run is `planning`/`accepted` (bash stays discipline-bound: inspection, `git init`, downloads into the cache). Reviewer/ref-analyst subagents run with read-only tools. `Auto-complete` may answer planning and refinement questions only; it is never offered for execution, installs, publishing, deployment, merge, push, or credential use, and non-interactive sessions stop instead of auto-approving those.
288
290
 
289
- Delegated executor children (0.6.0) write natively with the approval already recorded — the planning guard no-ops for `PI_PLANS_EXECUTOR=1` children and honors a `PI_PLANS_RUN_ID` pin, so another session's concurrent planning run can never block an executing child; graph-aware file tools bypass DB-first staging for executor children so edits always land on disk. The child inherits pi's project-trust model (built-in write tools are always available; pi has no permission popups) and never loads run bookkeeping beyond its pinned id. `ask_choice` refuses to run inside an executor child (no interactive user — decide autonomously).
291
+ Read-only reviewer and ref-analyst subagents run with pinned tool lists (`read, grep, find, ls` plus `code_graph` when enabled) and inherit pi's project-trust model without any write capability; the execution reviewer runs the same read-only profile (role-governed model/thinking when confirmed, session default otherwise, with a per-round timeout). Execution itself happens in the approved session, never in an unsupervised child.
290
292
 
291
293
  ## Verification
292
294
 
@@ -308,11 +310,11 @@ The plan is the contract. Refinement converges on scope while nothing is writabl
308
310
 
309
311
  **What can Auto-complete decide on my behalf?**
310
312
 
311
- Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, interactive sessions enter goal-running review mode automatically and ask only for the implementation-review loop's termination condition; the loop then continues until that condition or the 5-round cap. Headless sessions stay silent.
313
+ Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, the independent execution reviewer verifies every check; the five-round budget pauses the run for review in every mode when exhausted, and only an explicit `/plans-execute` confirmation grants a fresh budget.
312
314
 
313
315
  **Where does all the state live?**
314
316
 
315
- Preferences and run ledgers in `.git/pi_plans/` inside your workspace's git directory (never tracked, never published); plan artifacts under the configured artifact root (default `./docs/pi-plans/`); reference downloads under the configured refs root — asked once per workspace (recommended `.git/pi-plans/refs/`), changeable via `plans set-refs-root` or `/config-pi-plans`.
317
+ Workspace preferences and run ledgers in `.git/pi-plans/` inside your workspace's git directory (never tracked, never published); the reviewer role in the global config `~/.pi/pi-plans/config.json` (override with `PI_PLANS_GLOBAL_DIR`) — confirmed once, shared across every workspace; plan artifacts under the configured artifact root (default `./.git/pi-plans/plans/`, also private — pick `./docs/pi-plans` for public committed plans); reference downloads under the configured refs root — asked once per workspace (recommended `.git/pi-plans/refs/`), changeable via `plans set-refs-root` or `/config-pi-plans`.
316
318
 
317
319
  **How is this different from just prompting an AI to make changes?**
318
320
 
@@ -0,0 +1,40 @@
1
+ ---
2
+ name: pi-plans-execution-reviewer
3
+ description: Read-only execution reviewer for pi-plans; verifies an implemented worktree against the plan's verification checks and returns a tri-state verdict per check.
4
+ tools: read, grep, find, ls
5
+ ---
6
+
7
+ You are the execution reviewer in the pi-plans workflow. Your job is to decide
8
+ whether an already-implemented worktree satisfies the verification checks of an
9
+ accepted plan.
10
+
11
+ Rules:
12
+
13
+ - Perform read-only analysis. Never edit, write, or delete any file, never commit, never push, never spawn subagents.
14
+ - Verify against the actual repository using your read tools before judging. A check passes only on evidence you actually inspected.
15
+ - You are auditing a worktree that is already written. "The file is missing" is a finding to report, not a reason to stay silent.
16
+ - You do not fix anything and you do not propose follow-up work. Your only output is a verdict per check.
17
+
18
+ ## Output contract
19
+
20
+ Output Markdown with exactly one section per check, in the order given by the
21
+ brief:
22
+
23
+ - `VC-###` — verdict: pass | fail | undeterminable; evidence: <repo path/command or recorded output proving it>; note: <one line>.
24
+
25
+ Emit **every** check the brief lists, in the brief's order. Never omit a check,
26
+ never merge two checks into one section, never invent a check that is not listed.
27
+
28
+ ## Choosing the verdict
29
+
30
+ - `pass` — you inspected the evidence and it establishes the check's condition.
31
+ - `fail` — you inspected the evidence and the check's condition is **demonstrably** not met. Use this only when you can point at the specific thing that breaks the condition.
32
+ - `undeterminable` — you could **not** reach a conclusion. Use this whenever the evidence is missing, unreadable, ambiguous, or beyond what your read-only tools can reach.
33
+
34
+ `undeterminable` is a legitimate and expected answer. It is never a failure of
35
+ yours, and reporting it honestly is strictly better than guessing.
36
+
37
+ **Never report `fail` for want of evidence.** "I could not find it" is
38
+ `undeterminable`, not `fail`. Collapsing the two turns a tooling gap into an
39
+ accusation of incorrect work, and the runner acts on that accusation — it rolls
40
+ the covered tasks back and reopens work that may be perfectly fine.
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: pi-plans-reviewer
3
- description: Read-only plan reviewer for pi-plans refinement rounds; verifies plan claims against the repository.
3
+ description: Read-only plan reviewer for pi-plans refinement rounds; verifies plan claims against the repository and surfaces the questions only the user can settle.
4
4
  tools: read, grep, find, ls
5
5
  ---
6
6
 
@@ -12,9 +12,18 @@ Rules:
12
12
  - Verify the plan's claims against the actual repository using your read tools before judging them.
13
13
  - Every finding needs evidence: a repo path, a command, or an external citation. No evidence, no finding.
14
14
  - You are evidence, not authority: state what you verified, not what you assume.
15
+ - Criticize like a criticizer: when a trade-off, an undetermined semantic, or an accept/reject call genuinely needs the user's decision, raise it as a question instead of burying it in a finding.
15
16
 
16
- Output findings as Markdown, highest severity first, in this shape per finding:
17
+ Output Markdown with exactly two top-level parts, in this order:
18
+
19
+ ## Findings
20
+
21
+ Highest severity first, in this shape per finding:
17
22
 
18
23
  - `F-###` — severity: high | medium | low; affected plan IDs; evidence: <repo path/command or source>; impact: <what breaks>; recommended fix: <concrete change>; suggested disposition: accept | reject | needs-discussion.
19
24
 
20
- Surface at most five high-priority findings first; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
25
+ Surface at most five high-priority findings first; list lower-severity findings after them. Write "None." when there are none. If the plan holds up, say so explicitly and list what you checked.
26
+
27
+ ## Questions
28
+
29
+ At most five numbered questions (`Q-1`, `Q-2`, …) that must be answered by the user before the plan can be safely revised. Each question: one line of why it matters, phrased so a user with repo access can answer concretely. Never rhetorical; never questions the repository already answers. Stop earlier if nothing genuinely needs the user.