pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
@@ -0,0 +1,126 @@
1
+ # Contributing to pi-plans
2
+
3
+ Thanks for wanting to improve pi-plans. This guide covers the full path from a fresh clone to a merged pull request.
4
+
5
+ ## Expectations
6
+
7
+ pi-plans is maintained by a single person in their spare time. Review may take a few days, and not every pull request gets merged — that is normal. Small fixes (typos, docs, tests, one-file bug fixes) are welcome as direct pull requests. For larger changes — new concepts, public API changes, or workflow changes — please open an issue first so we can agree on the approach before you invest time.
8
+
9
+ ## Ways to contribute
10
+
11
+ You do not have to write code:
12
+
13
+ - Reproduce a bug and post exact steps, versions, and logs in an issue
14
+ - Fix or improve documentation (`README.md`, `references/`)
15
+ - Improve skill copy or prompts (`skills/`, `agents/`)
16
+ - Add or sharpen test cases (`tests/`)
17
+ - Verify behavior on your platform (macOS, Linux, different terminals, kitty/SSH)
18
+
19
+ Issues labeled `good first issue` are a good starting point.
20
+
21
+ ## Orientation
22
+
23
+ ```
24
+ pi-plans/
25
+ ├── index.ts # Extension entry: tools, commands, guard, execution loop
26
+ ├── tools/ # plans, ask-choice, refine, execute-plan, code-graph tools
27
+ ├── src/ # state, guard, plan parsing, subagent runner, refine UI, exec loop
28
+ │ └── code-graph/ # SQLite schema/store, parsers, indexer, summary, materialize
29
+ ├── skills/ # Planning router plus five specialist planning skills
30
+ ├── references/ # Shared workflow, state/config, plan template (normative)
31
+ ├── agents/ # reviewer.md subagent prompt (read-only; ref-analyst.md for reference analysis)
32
+ ├── scripts/ # validate.ts (structure + package artifact guard), run-tests.ts
33
+ └── tests/ # node:test suite
34
+ ```
35
+
36
+ `npm run validate` enforces several invariants, so keep them intact:
37
+
38
+ - every directory under `skills/` has a `SKILL.md` with frontmatter (`name` matching the directory, routing language in `description`) and the required phrases (`ask_choice`, `refine`, `.git/pi-plans`, ...)
39
+ - the `reviewer` agent prompt declares read-only tools and states the read-only contract
40
+ - the npm artifact stays code-sized: no `scripts/bench/vendor|results` entries, unpacked < 5 MiB, packed < 3 MiB, and key entries present
41
+ - `package.json` metadata (license, `pi-package` keyword, engines, scripts, required `files`) stays as asserted
42
+
43
+ ## Development setup
44
+
45
+ Requirements: Node.js >= 22.6 (the suite runs with `--experimental-strip-types`). The code-graph tests need `node:sqlite`: Node >= 22.13 unflagged, or `--experimental-sqlite` on 22.6–22.12.
46
+
47
+ If you don't have write access, fork the repository first and clone your fork.
48
+
49
+ ```bash
50
+ git clone https://github.com/MaxInGaussian/pi-plans
51
+ cd pi-plans
52
+ npm install # devDependencies only; nothing ships at runtime
53
+ npm run validate # structure + package artifact guard
54
+ npm test # node:test suite
55
+ ```
56
+
57
+ There are no runtime dependencies, but `npm install` is still required: source modules and tests resolve `@earendil-works/pi-*` (the code-graph tests additionally need `tree-sitter`) from `node_modules` at runtime.
58
+
59
+ To try the extension against a real session: `pi -e /path/to/pi-plans`.
60
+
61
+ ## Before you submit
62
+
63
+ Run both checks locally — CI runs exactly these:
64
+
65
+ ```bash
66
+ npm run validate
67
+ npm test
68
+ ```
69
+
70
+ Documentation duty: if your change alters behavior or the public API, update `README.md` or the relevant file under `references/` in the same pull request.
71
+
72
+ ## Commit messages
73
+
74
+ Follow the existing style: `type: short imperative description` (a scope is optional, e.g. `fix(form): ...`).
75
+
76
+ - `feat:` new feature
77
+ - `fix:` bug fix
78
+ - `docs:` documentation
79
+ - `refactor:` no behavior change
80
+ - `perf:` performance
81
+ - `test:` tests only
82
+ - `chore:` housekeeping
83
+ - `ci:` CI changes
84
+
85
+ Use the imperative mood ("fix race in ...", not "fixed ..."), keep the subject short, and use the body for motivation and evidence. Do not add generator or `Co-Authored-By` trailers.
86
+
87
+ ## Issues and pull requests
88
+
89
+ - Small fixes, docs, and tests: open the pull request directly.
90
+ - Larger changes (new concepts, public API or workflow changes): open an issue first and wait for a maintainer response.
91
+ - Bug fixes: reference the issue in the title, e.g. `fix: handle empty plan path (fix #12)`.
92
+ - Fill in the pull request template: Summary, Test Plan, and Docs sections.
93
+ - Keep one logical change per PR when you can; multiple small commits are fine (they are squashed on merge).
94
+
95
+ ## Tests
96
+
97
+ Every new behavior or bug fix must come with an executable test in `tests/` that fails without your change and passes with it. Tests run on `node:test` — no extra test framework is needed.
98
+
99
+ ## Dependencies
100
+
101
+ pi-plans ships zero runtime dependencies on purpose. Before adding a dependency:
102
+
103
+ - prefer a small, well-maintained package with no large transitive tree
104
+ - add it to `devDependencies` unless it is genuinely required at runtime
105
+ - explain in the pull request why it is needed and what you considered instead
106
+
107
+ ## Changelog
108
+
109
+ Do not edit `CHANGELOG.md` — maintainers write release entries when publishing.
110
+
111
+ ## AI and agents
112
+
113
+ Using AI assistance is fine, and most contributors here do. Two rules:
114
+
115
+ - You must understand your change: be able to explain what it does and how it interacts with the rest of the system. Pull requests that cannot be explained will be closed.
116
+ - Disclose AI assistance in the pull request (which tool, and to what extent). A real person must be behind every issue and PR; fully automated submissions with no human involvement may be closed.
117
+
118
+ Please write pull request descriptions and review replies yourself — short and specific beats long and generated.
119
+
120
+ ## Code of conduct
121
+
122
+ Be kind and constructive. This project follows the [Contributor Covenant](https://www.contributor-covenant.org/version/2/1/code_of_conduct/).
123
+
124
+ ## License
125
+
126
+ By contributing, you agree that your contributions are licensed under the [MIT License](LICENSE).
package/README.md CHANGED
@@ -22,7 +22,7 @@
22
22
 
23
23
  ---
24
24
 
25
- A rough change request becomes a versioned Markdown plan instead of a surprise diff. The agent inspects your repository read-only, asks scoped planning questions one at a time, and stores every answer in a per-run ledger. Reviewer and criticizer subagents refine the plan until it converges — and only after you explicitly approve the handoff does the extension enter a tracked execution loop that injects the remaining verifier checklist every turn and lifts the write guard. Nothing outside planning artifacts is writable until that approval.
25
+ A rough change request becomes a versioned Markdown plan instead of a surprise diff. The agent inspects your repository read-only, asks scoped planning questions one at a time, and stores every answer in a per-run ledger. Reviewer subagents refine the plan (findings plus questions) until it converges — and only after you explicitly approve the handoff does the extension enter a task-tree execution loop that injects the current wave and remaining tasks every turn, tracks progress through the `plans_update_task` tool, gates completion on an independent audit, and lifts the write guard. Nothing outside planning artifacts is writable until that approval.
26
26
 
27
27
  ## Benchmarked: 6× more tasks solved
28
28
 
@@ -51,6 +51,7 @@ at <b>7.6× fewer tokens per solved task</b>.
51
51
  - [Safety model](#safety-model)
52
52
  - [Verification](#verification)
53
53
  - [FAQ](#faq)
54
+ - [Contributing](#contributing)
54
55
  - [License](#license)
55
56
 
56
57
  ## How it works
@@ -61,29 +62,31 @@ at <b>7.6× fewer tokens per solved task</b>.
61
62
  language + docs location (once per workspace)
62
63
  |
63
64
  planning questions, one ask_choice at a time | write guard ON
64
- | | only .git/pi_plans/,
65
+ | | only .git/pi-plans/,
65
66
  v | run artifacts, cache
66
- PLAN_vN.md + Verifier Checklist | are writable
67
+ PLAN_vN.md (## Tasks + ## Verification
68
+ Checks) | are writable
67
69
  ^ |
68
70
  | refine rounds |
69
71
  +----------------+
70
- reviewer (x1..x3) -> criticizer -> revise
72
+ reviewer (x1..x3): findings + questions
71
73
  |
72
74
  v
73
75
  explicit approval (never auto-completed)
74
76
  |
75
77
  =============================================== write guard OFF
76
78
  |
77
- tracked execution loop
79
+ task-tree execution loop
78
80
  fused AGENTS.md × Ponytail executor rules
79
- checklist injected each turn, [DONE:VC-xxx]
80
- markers tracked via bottom status bar
81
+ current wave + tasks injected each turn,
82
+ plans_update_task reports status + evidence,
83
+ completion auditor verifies every check
81
84
  |
82
85
  v
83
86
  run status: done
84
87
  ```
85
88
 
86
- Every plan version carries stable IDs (`I-###`, `VC-###`) that never get recycled across revisions, so acceptance criteria survive refinement rounds intact.
89
+ Every plan version carries stable IDs (`Task-N`, `VC-###`) that never get recycled across revisions, so the task tree and verification checks survive refinement rounds intact.
87
90
 
88
91
  ## Quick start
89
92
 
@@ -101,12 +104,12 @@ Then describe a change from any repository:
101
104
  You: Create a plan to split the execution loop into smaller modules.
102
105
 
103
106
  Pi: Which planning docs location should this workspace use?
104
- 1. ./docs/pi-plans (recommended)
105
- 2. ./.git/pi_plans/plans
107
+ 1. ./.git/pi-plans/plans (recommended)
108
+ 2. ./docs/pi-plans
106
109
  3. Other
107
110
  4. Auto-complete
108
111
 
109
- Pi: Wrote ./docs/pi-plans/2026-08-26-split-execution-loop/PLAN_v1.md
112
+ Pi: Wrote ./.git/pi-plans/plans/2026-08-26-split-execution-loop/PLAN_v1.md
110
113
  Example verifier item:
111
114
  - [ ] `VC-001` covers `I-001`; pass condition: `npm test` passes;
112
115
  evidence: test output; metric: zero failing tests.
@@ -120,25 +123,26 @@ Pi: Accept the plan and execute it now?
120
123
  You: 1 — accept and execute.
121
124
  ```
122
125
 
123
- Planning artifacts live under `./docs/pi-plans/YYYY-MM-DD-<topic>/` by default (public, committed). Prefer `.git/pi_plans/plans` if you want them private to the repository.
126
+ Planning artifacts live under `./.git/pi-plans/plans/YYYY-MM-DD-<topic>/` by default — private to the repository, never tracked, never published. Choose `./docs/pi-plans` instead if you want the plans public and committed alongside the code.
124
127
 
125
128
  ## What it does
126
129
 
127
130
  | Capability | In short |
128
131
  |---|---|
129
132
  | Planning router + five specialist skills | Start with `/skill:planning` to route to the narrowest matching specialist (`plan-small` → `plan-big`, `debug-and-plan`, `plan-with-refs`) |
130
- | Choice prompts | `ask_choice`: recommended option first, answers auto-recorded per run; choosing Auto-complete enables recommendation-only answers for later eligible questions in the current planning run, with `/plans-autocomplete-stop` available to take back control. After execution completes, the continuation prompt enters goal-running mode by default: it asks only for the implementation-review loop's termination condition and then keeps refining until the loop ends or the cap is reached. |
131
- | Refinement rounds | Read-only reviewer/criticizer Pi subagents consolidate findings into the next plan version; delegated runs have standalone `Reviewer`/`Criticizer` progress overlays that close before the tool result returns; `analyze_refs` shows the same kind of overlay titled `Refs` while per-reference analysis subagents run |
132
- | Workspace state | Config, runs, decisions, refs, and subagent ledgers in `.git/pi_plans/` (git common dir) |
133
+ | Choice prompts | `ask_choice`: recommended option first, answers auto-recorded per run; every option you author states its advantage and its drawback as `✓ <advantage> / ✗ <drawback>` in the configured language, so the user can weigh each option before answering; choosing Auto-complete enables recommendation-only answers for later eligible questions in the current planning run, with `/plans-autocomplete-stop` available to take back control. |
134
+ | Refinement rounds | Read-only reviewer Pi subagents return findings (`F-###`) and up to five questions (`Q-1..Q-5`) in one round; the main agent asks every question with `ask_choice` and records the answers before revising. Delegated runs have a standalone `Reviewer` progress overlay; `analyze_refs` shows the same kind of overlay titled `Refs` while per-reference analysis subagents run |
135
+ | Workspace state | Config, runs, decisions, refs, and subagent ledgers in `.git/pi-plans/` (git common dir) |
133
136
  | VCC compact | Active planning/execution compaction uses deterministic, no-LLM VCC-style summaries when Pi core emits manual `/compact`, threshold, or overflow events. Summaries use five bracket sections plus a brief transcript, keep a smart recent tail, support `keep:N`, and write VCC details/stats without adding `/pi-vcc` commands. |
134
- | Visible Refiner overlay | Delegated reviewer/criticizer subagents surface as a named public overlay in the TUI — one `Reviewer`/`Criticizer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
135
- | Tracked execution | Checklist injected each turn; `[DONE:VC-xxx]` markers drive completion; implementation items report progress with `[I-xxx:implemented]` / `[I-xxx:validating]` markers; the bottom status bar shows lifecycle, `x/y` progress, elapsed time, and input/output token usage in real time |
136
- | Goal-wait continuation | In TUI/RPC, only a fully settled agent with unpassed VCs and no pending input or compaction gets one hidden wake carrying the latest checklist. Tool turns never queue reminders or consume guard rounds. The status bar shows progress; 3 no-progress cycles or 6 waiting cycles pause continuation. User interruption and final model errors also pause it. Only genuine user input or `/plans-execute` resumes; extension messages cannot. Print/JSON single-shot sessions track progress without automatic wakes. `/plans-stop` terminates execution |
137
- | Execution handoff | The accepted plan resumes in the current session model; no separate model selection is performed. |
138
- | Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current `I-###`, implementation IDs, and remaining `VC-###` checklist into the VCC sections. The old current-I proactive trigger and model-generated summary path are removed. |
137
+ | Visible Refiner overlay | Delegated reviewer subagents surface as a named public overlay in the TUI — one `Reviewer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
138
+ | Tracked execution | The current wave and remaining tasks are injected each turn; task progress is reported exclusively through the `plans_update_task` tool (status + evidence / skipReason, audit-only rollback); the task dashboard shows the tree live (compact aboveEditor widget, Ctrl+Shift+T expanded view with ✓/▸/~/· markers, width-adaptive); a stall watchdog pauses after three settled rounds without task-state change; the status bar shows lifecycle, `x/y` task progress, elapsed time, and token usage in real time |
139
+ | Multi-run workdirs (0.6.0) | Several pi sessions can plan concurrently in one workdir: the run registry derives from `runs/` (no shared pointer to race), each session binds to its run, and same-topic runs get suffixed artifact dirs. `/plans-abandon`, `/plans-execute`, and `/resume-plans` are binding-first and open a descriptive run-picker form when more than one candidate exists; `/plans` lists all runs (newest first, bound run marked) |
140
+ | Completion auditor | When every task reaches a terminal state, an independent read-only auditor verifies each `VC-###` check against the worktree; failed checks roll their covered tasks (children cascade, skipped reopen) back to pending and inject the audit report; three failed rounds pause for the user — auto-approve/headless terminates as `stopped` instead of hanging. Checks with all-skipped coverage pass; checks covering no task never audit |
141
+ | Execution handoff | The accepted plan executes in the current session after explicit approval (never auto-completed); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; 0.6.0 in-flight runs resume compatibly (delegated-executor orphans re-approve, paused executions rebuild from the task tree) |
142
+ | Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current task, task ids, and remaining `VC-###` checks into the VCC sections. Proactive triggers and model-generated summary paths are removed. |
139
143
  | Planning-phase compaction | During `run.status=planning` with no active execution, pi-plans maps active run, artifact directory, latest plan path from session entries, and observed current-I markers into the VCC sections. Without an active planning run, compaction returns to Pi core. Additionally, creating a new run (`plans start-run`) proactively requests one pre-plan VCC compaction and resumes planning with a hidden message (default on; `prePlanCompact:false` disables). |
140
144
  | Efficient executor prompt | Each turn, the executor is steered by a fused rule set — Marcos Hernanz's AGENTS.md principles × Ponytail minimalism: layered growth, simplest implementation, long-term architecture (no stopgaps), library discipline — so plans finish in fewer tokens and fewer detours |
141
- | Write guard | `edit`/`write` blocked outside planning artifacts while a run is active |
145
+ | Write guard | `edit`/`write` blocked outside planning artifacts while a planning run is active in the workdir; the guard prefers the session-bound run and lists the allowed roots on refusal |
142
146
 
143
147
  ## Interface overview
144
148
 
@@ -146,23 +150,23 @@ Planning artifacts live under `./docs/pi-plans/YYYY-MM-DD-<topic>/` by default (
146
150
  |---|---|
147
151
  | `plans` | State CLI: `init`, `show`, `set-language`, `set-artifact-root`, `set-refs-root`, `set-role`, `start-run`, `set-status`, `record-decision`, `record-ref`, `record-subagent`, `record-checkpoint` (state-machine-validated workflow transitions) |
148
152
  | `ask_choice` | Numbered choice prompt; `autoComplete: false` for the merged accept/execute question and external-state questions |
149
- | `refine` | Reviewer/criticizer round via standalone read-only subagents (`--mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` for both roles when the workspace has the code graph enabled); `target: "plan"` (default) reviews the plan, `target: "implementation"` reviews the implemented worktree against the plan; delegated TUI runs show one `Reviewer`/`Criticizer` overlay (78% width × 78% height, top-center, ≥72 cols) with per-lane transcript, follow-bottom scroll, Tab focus, and retention until `Esc`; `reviewers: 3` renders three equal-height panes; enforces role/model confirmation gates |
150
- | `analyze_refs` | plan-with-refs reference analysis: one independent read-only subagent per downloaded reference (cwd = the ref directory), reusing the reviewer role gates and the concurrent overlay (titled `Refs`); batches of at most 3 lanes run sequentially; returns structured per-reference sections for `REF_ANALYSIS.md` |
151
- | `execute_plan` | Execution handoff: re-confirms with the user and enters extension-managed execution mode |
152
- | `/plans` | Show config, active run, and execution progress |
153
- | `/config-pi-plans` | Re-ask workspace defaults for language, artifact root, refs root, code graph, reviewer mode/model, and criticizer mode/model |
154
- | `/resume-plans` | Resume the repository's working plan in the CURRENT session across restarts: unfinished planning (pending question + answered decisions), reviewing (round/lane state, successful outputs reused), execution (approval digest + HEAD + verified VCs), and implementation review (termination condition + round count). The unfinished active run wins; otherwise a unique candidate resumes directly and multiple candidates get a chooser. Linked worktrees share candidates; a cross-worktree resume confirms, copies artifacts without overwriting, and resets approval + VC validity (the termination condition survives, rounds restart at 0). An unchanged plan digest with a changed HEAD keeps the authorization but re-verifies old VCs first. Busy sessions and runs actively owned by a live process only notify — no queueing, no takeover. Interactive (TUI/RPC) only |
155
- | `/plans-execute [plan.md]` | Resume a paused active execution without losing verified progress; otherwise enter the explicit execution handoff (defaults to highest `PLAN_vN.md`) |
153
+ | `refine` | Reviewer round via standalone read-only subagents (`--mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` when the workspace has the code graph enabled): findings (`F-###`) and up to five questions (`Q-1..Q-5`) per lane; the caller must ask every question with `ask_choice` and record answers before revising; delegated TUI runs show one `Reviewer` overlay (78% width × 78% height, top-center, ≥72 cols) with per-lane transcript, follow-bottom scroll, Tab focus, and retention until `Esc`; `reviewers: 3` renders three equal-height panes; enforces the reviewer gates — first use pops native model + effort panels in TUI (menus on RPC, text guidance headless), persisted to the global reviewer config |
154
+ | `analyze_refs` | plan-with-refs reference analysis: one independent read-only subagent per downloaded reference (cwd = the ref directory), reusing the reviewer model confirmation from the global config (the mode is not consulted — analysis always spawns) and the concurrent overlay (titled `Refs`); batches of at most 3 lanes run sequentially; returns structured per-reference sections for `REF_ANALYSIS.md` |
155
+ | `execute_plan` | Execution handoff: re-confirms with the user (never auto-completed) and enters task-tree execution mode (`plans_update_task` progress, dashboard, completion auditor); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; picks the run via a descriptive form when several planned runs coexist |
156
+ | `/plans` | Show config, all runs (newest first, bound run marked, cap 50), and execution progress |
157
+ | `/config-pi-plans` | Re-ask workspace defaults for language, artifact root, refs root, and code graph, plus the reviewer mode/model (keep/change menu; native model + effort panels on change in TUI; current-session skips the model step) |
158
+ | `/resume-plans` | Resume a run in the CURRENT session across restarts: unfinished planning (pending question + answered decisions), reviewing (round/lane state, successful outputs reused), and execution (approval digest + HEAD + recorded task progress; 0.6.0 delegated-executor orphans re-approve, legacy implementation-review phases map to done). Binding-first: the session-bound resumable run resumes directly; a unique candidate goes direct; multiple candidates get a descriptive chooser. Linked worktrees share candidates; a cross-worktree resume confirms, copies artifacts without overwriting, and resets approval + task progress. An unchanged plan digest with a changed HEAD keeps the authorization but re-opens closed tasks. Busy sessions and actively owned runs only notify — no queueing, no takeover. Interactive (TUI/RPC) only |
159
+ | `/plans-execute [plan.md]` | Resume a paused active execution without losing task progress; otherwise enter the explicit execution handoff (run-picker form when several planned runs coexist; legacy plans get an upgrade notice) |
156
160
  | `/update-plan [plan.md] [reason…]` | Interrupt-and-refine: stops execution (if any), returns the run to planning, and directs the agent to revise the plan into `PLAN_vN+1.md` while preserving verified work |
157
161
  | `/plans-autocomplete-stop` | Stop the current run's Auto-complete mode and return later planning questions to normal interaction |
158
- | `/init-graph` | Build the code graph: tree-sitter function index + cross-file call/import edges (EXTRACTED vs INFERRED confidence) + label-propagation communities; writes `.git/pi_plans/graph/GRAPH_REPORT.md` (subsystems, god nodes, edge stats). Full rebuilds never touch files with staged edits (fail-closed pending guard) |
162
+ | `/init-graph` | Build the code graph: tree-sitter function index + cross-file call/import edges (EXTRACTED vs INFERRED confidence) + label-propagation communities; writes `.git/pi-plans/graph/GRAPH_REPORT.md` (subsystems, god nodes, edge stats). Full rebuilds never touch files with staged edits (fail-closed pending guard) |
159
163
  | `/update-graph` | Incrementally reindex changed files (shared path used by apply/final-commit triggers and the watcher) |
160
164
  | `/apply-graph` | Materialize DB-first staged edits to the worktree; auto-reindexes the materialized set afterward |
161
165
  | `/graph-status` `/graph-drift` | Graph inventory and DB↔source convergence |
162
166
  | `/watch-graph` `/unwatch-graph` | 300ms-debounced filesystem watcher feeding incremental reindex; single-writer per worktree (atomically claimed PID+heartbeat lock, stale takeover), skips files with staged edits, fails closed on errors, stops cleanly on fs.watch errors / repeated failures / session shutdown, auto-restarts when enabled |
163
167
  | `code_graph` actions | Read-only: `status` (files/functions/edges + confidence×resolution distribution), `screening` (per-row freshness probes), `get-function`, `query` (keyword→BFS/DFS with token budget + shown/omitted counts), `path A B` (shortest call path), `explain <fn>` (community/degrees/neighbors), `impact <fn>` (reverse call closure + affected files); write path: DB-first staged edits + `apply` (auto-reindexes the materialized set). Edge resolution covers re-export barrels (`export {x} from` / `export *` follow to the defining module), namespace/default imports, `require` destructuring, and dynamic `import()`; same-name exports classify as `ambiguous` |
164
168
  | `/plans-stop` | Stop execution mode |
165
- | `/plans-abandon` | Abandon the active run (lifts the write guard; artifacts stay) |
169
+ | `/plans-abandon` | Abandon a run (pick via form when several candidates; lifts the write guard; artifacts stay) |
166
170
  | Status bar (lifecycle) | 💬 Q&A → 📝 draft written (planning sub-phases) → ⌛ executing `x/y · spent · in/out-toks` in the bottom status bar → ⛔ stopped / 🎯 done / 🚫 abandoned |
167
171
 
168
172
  ## VCC compact
@@ -175,11 +179,11 @@ Pi core remains the owner of compaction scheduling: manual `/compact`, threshold
175
179
  - **Manual matrix.** Plain `/compact` and `/compact keep:N` compact and show stats without continuing. `/compact <text>` and `/compact keep:N <text>` compact, then send the text once as the follow-up prompt. Internal pi-plans compaction markers are never reused as user follow-up prompts.
176
180
  - **Fallbacks and stats.** Unsafe manual/threshold cuts cancel with a warning; overflow or retrying unsafe cuts return control to Pi core. Successful VCC compactions notify with kept-tail and summarized-message stats. Threshold/overflow compactions may queue one hidden continuation only when the running Pi version still needs it and `continueAfterThresholdCompact` is enabled.
177
181
  - **Pre-plan compaction.** When `plans start-run` creates a new planning run, pi-plans proactively requests one VCC compaction (internal hint `pi-plans planning pre-plan compact`) right after the run is created and before the first planning question, then resumes the planning turn with a hidden message — so each new plan starts on a lean context (LLM reasoning degrades with longer input). Small sessions, already-compacted sessions, and failures skip silently and still resume. `prePlanCompact:false` in the repo-private config restores the old behavior.
178
- - **Repo-private config.** Defaults are scaffolded in `.git/pi_plans/pi-vcc-config.json` under the resolved git common dir: `overrideDefaultCompaction:true`, `smartKeepTail:true`, `continueAfterThresholdCompact:true`, `prePlanCompact:true`, `debug:false`. Global pi-vcc config and `PI_VCC_CONFIG_PATH` are intentionally ignored.
182
+ - **Repo-private config.** Defaults are scaffolded in `.git/pi-plans/pi-vcc-config.json` under the resolved git common dir: `overrideDefaultCompaction:true`, `smartKeepTail:true`, `continueAfterThresholdCompact:true`, `prePlanCompact:true`, `debug:false`. Global pi-vcc config and `PI_VCC_CONFIG_PATH` are intentionally ignored.
179
183
 
180
184
  ## Visible Refiner overlay
181
185
 
182
- Delegated `refine` rounds (reviewer or criticizer) and `analyze_refs` rounds (titled `Refs`) show their progress directly inside the Pi TUI instead of disappearing into the child process's terminal. The overlay is a public, named panel so users always know who is doing what:
186
+ Delegated `refine` rounds (reviewer) and `analyze_refs` rounds (titled `Refs`) show their progress directly inside the Pi TUI instead of disappearing into the child process's terminal. The overlay is a public, named panel so users always know who is doing what:
183
187
 
184
188
  - **Pi-btw-aligned geometry.** Each round uses `width: "78%"`, `minWidth: 72`, `maxHeight: "78%"`, `anchor: "top-center"`, and `{ top: 1, left: 2, right: 2 }` margins (no dependency on `pi-btw`; the renderer is built on Pi's public `pi-tui` primitives).
185
189
  - **Complete streaming transcript.** Assistant text, thinking blocks, tool calls, tool results, and stderr are merged per turn/content block into lane entries without overlay-facing truncation; only the viewport slices them. Final `message_end` / `tool_execution_end` overwrite the live snapshot with the authoritative content.
@@ -214,11 +218,11 @@ Invoked via `resources_discover`, callable as `/skill:<name>`, directly as `/<na
214
218
  | Skill | Use it when |
215
219
  |---|---|
216
220
  | [`planning`](skills/planning/SKILL.md) | General router; selects the narrowest specialist skill before planning starts |
217
- | [`plan-small`](skills/plan-small/SKILL.md) | Small scoped change; 1–3 questions; one criticizer round |
218
- | [`plan-normal`](skills/plan-normal/SKILL.md) | Broad or risky change; 5–10 questions; reviewer + criticizer rounds |
221
+ | [`plan-small`](skills/plan-small/SKILL.md) | Small scoped change; 1–3 questions; one reviewer round |
222
+ | [`plan-normal`](skills/plan-normal/SKILL.md) | Broad or risky change; 5–10 questions; reviewer rounds |
219
223
  | [`plan-big`](skills/plan-big/SKILL.md) | Open-ended/high-risk effort; 10+ questions; three concurrent reviewers |
220
224
  | [`debug-and-plan`](skills/debug-and-plan/SKILL.md) | Bug, CI failure, regression, incident — diagnose before planning |
221
- | [`plan-with-refs`](skills/plan-with-refs/SKILL.md) | External projects/papers/docs must be analyzed before planning |
225
+ | [`plan-with-refs`](skills/plan-with-refs/SKILL.md) | External references must be analyzed before planning — repos, papers (arXiv), engineering blogs, and docs sites all count; theoretical references are equal citizens. plan-normal/plan-big may optionally cite 1–2 search-found references without downloading |
222
226
 
223
227
  ## Installation details
224
228
 
@@ -275,14 +279,16 @@ pi-plans/
275
279
  │ └── code-graph/ # SQLite schema/store, parsers, indexer, summary, materialize
276
280
  ├── skills/ # The planning router plus five specialist planning skills
277
281
  ├── references/ # Shared workflow, state/config, plan template (normative)
278
- ├── agents/ # reviewer.md / criticizer.md subagent prompts
282
+ ├── agents/ # reviewer.md subagent prompt (ref-analyst.md for reference analysis)
279
283
  ├── scripts/validate.ts # Structure + package artifact guard
280
284
  └── tests/ # node:test suite (state, guard, plan parsing, execution, refine progress, code-graph)
281
285
  ```
282
286
 
283
287
  ## Safety model
284
288
 
285
- Before the approved handoff the workflow writes only `.git/pi_plans/` state, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root (set via `plans set-refs-root` or `/config-pi-plans`; the recommended `.git/pi-plans/refs/` lives inside the git dir and needs no extra guard) — the extension blocks `edit`/`write` elsewhere while a run is `planning`/`accepted` (bash stays discipline-bound: inspection, `git init`, downloads into the cache). Reviewer/criticizer/ref-analyst subagents run with read-only tools. `Auto-complete` may answer planning and refinement questions only; it is never offered for execution, installs, publishing, deployment, merge, push, or credential use, and non-interactive sessions stop instead of auto-approving those.
289
+ Before the approved handoff the workflow writes only `.git/pi-plans/` state, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root (set via `plans set-refs-root` or `/config-pi-plans`; the recommended `.git/pi-plans/refs/` lives inside the git dir and needs no extra guard) — the extension blocks `edit`/`write` elsewhere while a run is `planning`/`accepted` (bash stays discipline-bound: inspection, `git init`, downloads into the cache). Reviewer/ref-analyst subagents run with read-only tools. `Auto-complete` may answer planning and refinement questions only; it is never offered for execution, installs, publishing, deployment, merge, push, or credential use, and non-interactive sessions stop instead of auto-approving those.
290
+
291
+ Read-only reviewer and ref-analyst subagents run with pinned tool lists (`read, grep, find, ls` plus `code_graph` when enabled) and inherit pi's project-trust model without any write capability; the completion auditor runs the same read-only profile. Execution itself happens in the approved session, never in an unsupervised child.
286
292
 
287
293
  ## Verification
288
294
 
@@ -304,11 +310,11 @@ The plan is the contract. Refinement converges on scope while nothing is writabl
304
310
 
305
311
  **What can Auto-complete decide on my behalf?**
306
312
 
307
- Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, interactive sessions enter goal-running review mode automatically and ask only for the implementation-review loop's termination condition; the loop then continues until that condition or the 5-round cap. Headless sessions stay silent.
313
+ Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, the independent completion auditor verifies every check; interactive sessions pause for review if the audit exhausts its three rounds, while headless runs terminate bounded instead of hanging.
308
314
 
309
315
  **Where does all the state live?**
310
316
 
311
- Preferences and run ledgers in `.git/pi_plans/` inside your workspace's git directory (never tracked, never published); plan artifacts under the configured artifact root (default `./docs/pi-plans/`); reference downloads under the configured refs root — asked once per workspace (recommended `.git/pi-plans/refs/`), changeable via `plans set-refs-root` or `/config-pi-plans`.
317
+ Workspace preferences and run ledgers in `.git/pi-plans/` inside your workspace's git directory (never tracked, never published); the reviewer role in the global config `~/.pi/pi-plans/config.json` (override with `PI_PLANS_GLOBAL_DIR`) — confirmed once, shared across every workspace; plan artifacts under the configured artifact root (default `./.git/pi-plans/plans/`, also private — pick `./docs/pi-plans` for public committed plans); reference downloads under the configured refs root — asked once per workspace (recommended `.git/pi-plans/refs/`), changeable via `plans set-refs-root` or `/config-pi-plans`.
312
318
 
313
319
  **How is this different from just prompting an AI to make changes?**
314
320
 
@@ -318,6 +324,10 @@ Prompts produce one-shot diffs with no recorded reasoning. pi-plans produces ver
318
324
 
319
325
  The injected rule set is four compressed lines. It buys back more than it costs: the executor stops re-deriving discipline (no speculative abstractions, no compatibility detours, no reinvented helpers), so finished items converge in fewer turns and fewer tokens overall.
320
326
 
327
+ ## Contributing
328
+
329
+ Contributions are welcome — see [CONTRIBUTING.md](CONTRIBUTING.md) for the full workflow (setup, checks, commit style, and PR expectations). Small fixes can go straight to a pull request; for larger changes, open an issue first. Look for issues labeled `good first issue` to get started.
330
+
321
331
  ## License
322
332
 
323
333
  MIT.
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: pi-plans-ref-analyst
3
- description: Read-only reference analyst for pi-plans plan-with-refs runs; deep-reads one downloaded reference and extracts adoptable ideas for the target repository.
3
+ description: Read-only reference analyst for pi-plans plan-with-refs runs; deep-reads ONE downloaded reference of any medium (repo, paper, blog, docs) and extracts adoptable ideas for the target repository.
4
4
  tools: read, grep, find, ls
5
5
  ---
6
6
 
@@ -9,9 +9,12 @@ You are a read-only reference analyst in the pi-plans plan-with-refs workflow.
9
9
  Rules:
10
10
 
11
11
  - Perform read-only analysis. Never edit, write, or delete any file.
12
- - Your working directory is the local copy of ONE downloaded reference. Deep-read it: entry points, README/docs, core modules, tests, and configuration.
13
- - Judge the reference through the lens of the target repository described in your task: what is worth borrowing, what is not, and why.
14
- - Every claim needs evidence: a file path inside the reference (with line numbers when quoting). No evidence, no claim.
12
+ - Your working directory is the local copy of ONE downloaded reference. Deep-read it according to its medium:
13
+ - code repository: entry points, README/docs, core modules, tests, and configuration;
14
+ - paper: claims, method, limitations, and experiments/measurements;
15
+ - blog post or docs site: the technique, the measurements/benchmarks, and the caveats.
16
+ - Judge the reference through the lens of the target repository described in your task: what is worth borrowing, what is not, and why. Theoretical grounding counts — an algorithm, a formal property, or a measured tradeoff is as adoptable as an implementation pattern.
17
+ - Every claim needs evidence. For code: a file path inside the reference (with line numbers when quoting). For papers: section/theorem/table numbers with a short quote. For blogs/docs: the heading or quoted passage. No evidence, no claim.
15
18
  - You are evidence, not authority: state what you verified, not what you assume.
16
19
  - Stay inside the reference directory; do not wander the filesystem.
17
20
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: pi-plans-reviewer
3
- description: Read-only plan reviewer for pi-plans refinement rounds; verifies plan claims against the repository.
3
+ description: Read-only plan reviewer for pi-plans refinement rounds; verifies plan claims against the repository and surfaces the questions only the user can settle.
4
4
  tools: read, grep, find, ls
5
5
  ---
6
6
 
@@ -12,9 +12,18 @@ Rules:
12
12
  - Verify the plan's claims against the actual repository using your read tools before judging them.
13
13
  - Every finding needs evidence: a repo path, a command, or an external citation. No evidence, no finding.
14
14
  - You are evidence, not authority: state what you verified, not what you assume.
15
+ - Criticize like a criticizer: when a trade-off, an undetermined semantic, or an accept/reject call genuinely needs the user's decision, raise it as a question instead of burying it in a finding.
15
16
 
16
- Output findings as Markdown, highest severity first, in this shape per finding:
17
+ Output Markdown with exactly two top-level parts, in this order:
18
+
19
+ ## Findings
20
+
21
+ Highest severity first, in this shape per finding:
17
22
 
18
23
  - `F-###` — severity: high | medium | low; affected plan IDs; evidence: <repo path/command or source>; impact: <what breaks>; recommended fix: <concrete change>; suggested disposition: accept | reject | needs-discussion.
19
24
 
20
- Surface at most five high-priority findings first; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
25
+ Surface at most five high-priority findings first; list lower-severity findings after them. Write "None." when there are none. If the plan holds up, say so explicitly and list what you checked.
26
+
27
+ ## Questions
28
+
29
+ At most five numbered questions (`Q-1`, `Q-2`, …) that must be answered by the user before the plan can be safely revised. Each question: one line of why it matters, phrased so a user with repo access can answer concretely. Never rhetorical; never questions the repository already answers. Stop earlier if nothing genuinely needs the user.