pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
@@ -9,17 +9,17 @@ This skill set is written for the Pi coding agent's documented behavior:
9
9
  - the five skills are contributed by the pi-plans extension and loaded as Pi skills (also invokable as `/skill:<name>`);
10
10
  - skill references and helper sources are resolved relative to the directory containing `SKILL.md`;
11
11
  - Planning and reference analysis run with the extension tools `plans`, `ask_choice`, `refine`, `analyze_refs`, and `execute_plan`;
12
- - `refine` spawns read-only Pi subagents (`pi --mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` for both roles when workspace `graph_enabled` is true) with isolated context; delegated Reviewer/Criticizer runs show a standalone aggregate overlay titled `Reviewer` or `Criticizer` (78% × 78% top-center, ≥72 cols, no input row), stream assistant/thinking/tool events into per-lane transcripts with follow-bottom scroll, dismiss on `Esc` (close-only — the refiner child keeps running and its result still flows back as tool output), replace any retained finished overlay when a new round begins, and return conclusions to the main session as tool output; `analyze_refs` spawns one read-only subagent per downloaded reference (cwd = that ref's directory) reusing the reviewer role gates, shows the same overlay titled `Refs` in batches of at most 3 lanes, and returns structured per-reference sections for `REF_ANALYSIS.md`;
12
+ - `refine` spawns read-only Pi subagents (`pi --mode json -p --no-session --tools read,grep,find,ls`, plus `code_graph` when workspace `graph_enabled` is true) with isolated context; delegated reviewer runs show a standalone aggregate overlay titled `Reviewer` (78% × 78% top-center, ≥72 cols, no input row), stream assistant/thinking/tool events into per-lane transcripts with follow-bottom scroll, dismiss on `Esc` (close-only — the refiner child keeps running and its result still flows back as tool output), replace any retained finished overlay when a new round begins, and return conclusions to the main session as tool output; `analyze_refs` spawns one read-only subagent per downloaded reference (cwd = that ref's directory) reusing the reviewer role gates, shows the same overlay titled `Refs` in batches of at most 3 lanes, and returns structured per-reference sections for `REF_ANALYSIS.md`;
13
13
  - when graph mode is enabled, graph-aware `read`/`edit` overrides are active for indexed source files: `read` returns a capped function digest (≤50 lines, synthetic anonymous entries folded) by default — drill in via `offset/limit` or `code_graph get-function`, and `full: true` is the only whole-file exit (small/zero-function files return full text; safety truncation matches native read); `write`/`edit` stage DB-first mutations until materialized via the `code_graph` tool's `apply` action (same planning/accepted gate as /apply-graph; refused for read-only refiner subagents via the PI_PLANS_REFINER marker; returns a per-file report with counts and a post-apply drift summary, and never changes run status); unexpected fallbacks (`not indexed` / `runtime unavailable` / `config read failed`) are marked at the top of the result while flag-off fallbacks stay unmarked;
14
- - the execution loop is extension-managed: remaining verifier items are injected each turn, implementation items emit `[I-###:current]`/`[I-###:implemented|validating]`, and `[DONE:VC-xxx]` markers are tracked with a bottom status bar;
14
+ - the execution loop is extension-managed and task-tree driven: the current wave and remaining tasks are injected each turn, progress is reported exclusively through the `plans_update_task` tool (status + evidence / skipReason), the task dashboard tracks every task (compact widget; Ctrl+Shift+T expands the tree), and an independent completion auditor verifies the verification checks before the run completes;
15
15
  - execution and planning compaction keep Pi's SessionManager as the history owner; during active pi-plans runs, `session_before_compact` uses a deterministic no-LLM VCC-style summary with `[Session Goal]`, `[Files And Changes]`, `[Commits]`, `[Outstanding Context]`, `[User Preferences]`, and a ranked brief transcript; Pi core owns manual `/compact`, threshold, and overflow scheduling, while pi-plans handles smart tail keep, `keep:N`, stats, and phase-specific run/plan/current-I/checklist context; in addition, creating a new planning run (`plans start-run`) proactively requests one pre-plan VCC compaction before the first planning question and resumes the planning turn with a hidden message (default on, `prePlanCompact` in `pi-vcc-config.json`);
16
16
 
17
17
  ## Planning Boundary
18
18
 
19
19
  - Treat the user's request as a planning target, not as write authorization.
20
- - Before the execution handoff, do not edit target source files, docs, configs, package metadata, generated assets, or tests outside the planning artifact directory and the pi-plans state under `.git/pi_plans/`. The extension enforces this for `edit` and `write` while a run is active: only `.git/pi_plans/`, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root are writable. Bash is not machine-guarded — keep it read-only by discipline (inspection, `git init`, downloads into the cache).
21
- - The normal pre-handoff writes are `.git/pi_plans/` state plus planning artifacts under the configured artifact root (default `./docs/pi-plans/...`).
22
- - Downloaded references go to the workspace's configured refs root (`refs_root` in `.git/pi_plans/config.json`; unset → ask once via `ask_choice`, recommended `.git/pi-plans/refs/`, second `./refs/`, third `~/.cache/pi-plans/refs/`; persist with the `plans` tool, `set-refs-root`) and their paths and evidence are recorded in `REF_ANALYSIS.md`. References are not limited to GitHub repositories: papers (arXiv etc.), engineering blogs, and documentation sites are first-class, with per-medium qualified downloads (repo = clone; paper = full text — abstracts never qualify; blog/docs = full-article markdown) and one directory per reference; theoretical references count exactly as much as implementation references. plan-normal and plan-big may optionally cite 1–2 search-found references (URLs in the plan's Evidence section) without downloading them.
20
+ - Before the execution handoff, do not edit target source files, docs, configs, package metadata, generated assets, or tests outside the planning artifact directory and the pi-plans state under `.git/pi-plans/`. The extension enforces this for `edit` and `write` while a run is active: only `.git/pi-plans/`, the run's artifact directory, `~/.cache/pi-plans/`, and the configured refs root are writable. Bash is not machine-guarded — keep it read-only by discipline (inspection, `git init`, downloads into the cache).
21
+ - The normal pre-handoff writes are `.git/pi-plans/` state plus planning artifacts under the configured artifact root (default `./.git/pi-plans/plans/...`).
22
+ - Downloaded references go to the workspace's configured refs root (`refs_root` in `.git/pi-plans/config.json`; unset → ask once via `ask_choice`, recommended `.git/pi-plans/refs/`, second `./refs/`, third `~/.cache/pi-plans/refs/`; persist with the `plans` tool, `set-refs-root`) and their paths and evidence are recorded in `REF_ANALYSIS.md`. References are not limited to GitHub repositories: papers (arXiv etc.), engineering blogs, and documentation sites are first-class, with per-medium qualified downloads (repo = clone; paper = full text — abstracts never qualify; blog/docs = full-article markdown) and one directory per reference; theoretical references count exactly as much as implementation references. plan-normal and plan-big may optionally cite 1–2 search-found references (URLs in the plan's Evidence section) without downloading them.
23
23
  - After the user explicitly approves the execution handoff, leave this planning workflow and execute in the extension-managed loop (see Execution Handoff).
24
24
 
25
25
  ## State And Settings
@@ -30,13 +30,13 @@ Before the first planning question, read `references/state-and-config.md` and in
30
30
  { "action": "init", "workdir": "<target-workdir>" }
31
31
  ```
32
32
 
33
- State lives under the workspace's resolved git common dir as `.git/pi_plans/` (auto-ignored, no `.gitignore` entries; the tool auto-runs `git init` when the workdir safely has no repository). The target workspace is the current working directory unless the user explicitly names another repository.
33
+ State lives under the workspace's resolved git common dir as `.git/pi-plans/` (auto-ignored, no `.gitignore` entries; the tool auto-runs `git init` when the workdir safely has no repository). The target workspace is the current working directory unless the user explicitly names another repository.
34
34
 
35
- If `language.tag` is missing from `.git/pi_plans/config.json`, ask the language setting question (via `ask_choice`) before any product question. Persist it with `plans` (`set-language`); this question does not count against the planning-question limit.
35
+ If `language.tag` is missing from `.git/pi-plans/config.json`, ask the language setting question (via `ask_choice`) before any product question. Persist it with `plans` (`set-language`); this question does not count against the planning-question limit.
36
36
 
37
- If `artifact_root_source` is missing from `.git/pi_plans/config.json` or is `unset`, ask the planning docs location question (via `ask_choice`) before any product question. Persist it with `plans` (`set-artifact-root`); this question does not count against the planning-question limit.
37
+ If `artifact_root_source` is missing from `.git/pi-plans/config.json` or is `unset`, ask the planning docs location question (via `ask_choice`) before any product question. Persist it with `plans` (`set-artifact-root`); this question does not count against the planning-question limit.
38
38
 
39
- Reviewer and criticizer settings also live in `config.json`. If a role's mode is missing/invalid or its `confirmed_at` is `null` when the role is about to run, ask the matching role-setting or model-confirmation question (via `ask_choice`), persist with `plans` (`set-role`), and then run the `refine` tool. The `refine` tool refuses to spawn until the gates pass.
39
+ The reviewer role lives in the GLOBAL config (`~/.pi/pi-plans/config.json`, `PI_PLANS_GLOBAL_DIR` override — one confirmation for every workspace). If its mode is missing/invalid when a round is about to run, ask the role-setting question via `ask_choice` and persist with `plans` (`set-role`). The model + thinking level are confirmed at first use through NATIVE panels, not ask_choice: in TUI the `refine` gate itself pops a searchable model panel followed by an effort panel (Default row = no `--thinking` flag; other rows follow the chosen model's `thinkingLevelMap`) and continues the same invocation on completion; hasUI non-TUI sessions get native menus; UI-less sessions get text guidance embedding the available selectors. Esc cancels the whole gate (nothing persisted; do NOT re-ask via ask_choice — suggest `/config-pi-plans`). The `refine` tool refuses to spawn until `reviewerReady` passes (current-session, or delegated with a confirmed concrete `provider/model`). (v0.6.1: the criticizer role is gone — the reviewer emits findings AND questions in one round.)
40
40
 
41
41
  ## First-Turn Contract
42
42
 
@@ -59,7 +59,7 @@ When the recommended option depends on a web-verifiable claim, search first (web
59
59
 
60
60
  Every user-facing planning or refinement question goes through the `ask_choice` tool:
61
61
 
62
- - `options`: ordered options, recommended option first with `recommended: true` (exactly one), each with a `description` of the form `✓ <advantage> / ✗ <drawback>` — **every option you author states both what it gains and what it costs**, in the configured language, tersely (≈8 words per half; write `—` when a side is genuinely absent). Both halves live in the single `description` string; there are no separate `pros`/`cons` fields. This rule covers every AI-written option, including the scope confirmation, the merged accept/execute handoff, and the implementation-review setup questions; only the tool-appended `Other…` / `Auto-complete` / `Auto-refine loop` rows are exempt;
62
+ - `options`: ordered options, recommended option first with `recommended: true` (exactly one), each with a `description` of the form `✓ <advantage> / ✗ <drawback>` — **every option you author states both what it gains and what it costs**, in the configured language, tersely (≈8 words per half; write `—` when a side is genuinely absent). Both halves live in the single `description` string; there are no separate `pros`/`cons` fields. This rule covers every AI-written option, including the scope confirmation and the merged accept/execute handoff; only the tool-appended `Other…` / `Auto-complete` rows are exempt;
63
63
  - do not add `Other` or `Auto-complete` yourself — the tool appends `Other…` second-last and `Auto-complete` last;
64
64
  - pass `autoComplete: false` for the merged accept/execute question — it contains the execution approval, so Auto-complete never appears there — and for any install waiver, publishing, deployment, merge, push, credential, or external-state question. Auto-complete may choose the recommended planning or refinement option only.
65
65
  - When the user selects Auto-complete, it remains active for the current planning run: later eligible questions use their recommended options automatically, and the extension queues one deduplicated follow-up if the model stops after an auto-completed answer. `/plans-autocomplete-stop` disables it; session restore may reactivate it only for the same active run while its status is `planning`.
@@ -74,9 +74,9 @@ Create a run only after initial read-only inspection makes the topic clear:
74
74
  { "action": "start-run", "workdir": "<target-workdir>", "topic": "<short topic>", "skill": "<skill-name>", "requestText": "<original request>" }
75
75
  ```
76
76
 
77
- The tool creates the configured artifact directory root (default `./docs/pi-plans/YYYY-MM-DD-<topic>/`) and the private run state `.git/pi_plans/runs/<run-id>/`, and stores the active run pointer. Use a short lowercase slug for the topic. Keep paths stable once written.
77
+ The tool creates the configured artifact directory root (default `./.git/pi-plans/plans/YYYY-MM-DD-<topic>/`) and the private run state `.git/pi-plans/runs/<run-id>/`, and stores the active run pointer. Use a short lowercase slug for the topic. Keep paths stable once written.
78
78
 
79
- `DECISIONS.md` records: original request; repository evidence inspected; each question, options, selected answer, and whether it came from the user or Auto-complete; assumptions still open; external sources consulted; language, reviewer, and criticizer settings used.
79
+ `DECISIONS.md` records: original request; repository evidence inspected; each question, options, selected answer, and whether it came from the user or Auto-complete; assumptions still open; external sources consulted; language and reviewer settings used.
80
80
 
81
81
  ## Final Scope Confirmation
82
82
 
@@ -91,15 +91,15 @@ If the user adds requirements, resolve only the necessary follow-up questions, t
91
91
 
92
92
  ## Plan Artifact Requirements
93
93
 
94
- Every `PLAN_vN.md` must include stable IDs that are never recycled across revisions: goals and non-goals; requirements and constraints; implementation items; affected paths; dependencies and sequencing; risks and mitigations; acceptance criteria; verification steps; repo and external evidence; resolved decisions; revision ledger.
94
+ Every `PLAN_vN.md` must include stable IDs that are never recycled across revisions: `Task-N` tasks (with deps/files/wave and one subtask level), `VC-###` verification checks, and the revision ledger. Everything else the workflow needs lives in the run state (decisions ledger, review rounds, refs), not in the plan file.
95
95
 
96
- Every plan version must include a dedicated `## Verifier Checklist` section. Each item is a Markdown checkbox of the exact shape:
96
+ Every plan version's body is exactly two sections — `## Tasks` and `## Verification Checks` — plus the metadata header (see `references/plan-artifact-template.md` for the full microsyntax: `Task-N` ids, one subtask level, inline `deps:`/`files:`/`wave:` fields, the `### Execution Waves` subsection, and lint rules). Each verification check is a Markdown checkbox of the exact shape:
97
97
 
98
98
  ```markdown
99
- - [ ] `VC-001` covers `I-001`; pass condition: ...; evidence: ...; metric: <threshold or reason not quantified>.
99
+ - [ ] `VC-001` covers `Task-1`; pass condition: ...; evidence: ...; metric: <threshold or reason not quantified>.
100
100
  ```
101
101
 
102
- The execution loop parses `- [ ] \`VC-###\`` items and tracks `[DONE:VC-###]` markers, so keep IDs on the checkbox line. Use `references/plan-artifact-template.md` when drafting.
102
+ The execution loop parses `- [ ] \`VC-###\`` checks and the task tree, so keep IDs on the checkbox line and the task grammar exact; task progress flows through `plans_update_task` and the completion auditor reads these checks. Use `references/plan-artifact-template.md` when drafting.
103
103
 
104
104
  ## Refinement
105
105
 
@@ -109,78 +109,62 @@ After each plan version, ask one merged accept/execute question via `ask_choice`
109
109
  2. `Accept PLAN_vN, don't execute yet` — mark accepted; resume later via `/plans-execute`.
110
110
  3. `Run another round: <the level's default next refine mode>` — only while the level's default sequence is unfinished.
111
111
 
112
- The recommended option follows the skill level's default sequence: while the default rounds are unfinished it is option 3's default next mode (`plan-small`: `Criticizer`; `plan-normal`: `Reviewer` then `Criticizer`; `plan-big`: three concurrent reviewers (`refine` with `reviewers: 3`) then `Criticizer`); once the default sequence is complete it is option 1.
112
+ The recommended option follows the skill level's default sequence: while the default round is unfinished it is option 3 (`plan-small` / `plan-normal`: one reviewer round; `plan-big` / `plan-with-refs`: three concurrent reviewers via `refine` with `reviewers: 3`); once the default round is complete it is option 1. Every round returns findings (`F-###`) and up to five questions (`Q-1..Q-5`) in the same output.
113
113
 
114
- If the user selects `Reviewer` or `Criticizer`, run the `refine` tool with the plan path and any focus. Reviewer output consolidates into `PLAN_vN_reviewer_comments.md` with findings IDs, severity, affected plan IDs, evidence, impact, recommended fix, and disposition. Revise the next plan only for findings accepted on evidence.
114
+ If the user selects another round, run the `refine` tool with the plan path and any focus. Reviewer output consolidates into `PLAN_vN_reviewer_comments.md` with findings IDs, severity, affected plan IDs, evidence, impact, recommended fix, and disposition. Revise the next plan only for findings accepted on evidence.
115
115
 
116
116
  ### Concurrent Reviewers (big plans)
117
117
 
118
118
  A big-plan reviewer round runs three independent reviewer subagents (`reviewers: 3`); each gets its own emphasis lens but forms its own priorities. After they return, merge and dedupe their findings into one consolidated `PLAN_vN_reviewer_comments.md`, keeping each finding's source reviewer, severity, evidence, and disposition, and surface at most five high-priority comments to the user. Treat agreement between independent reviewers as stronger evidence, not as authority; every accepted finding still needs repo or reference evidence.
119
119
 
120
- ### Criticizer Rounds
120
+ ### Reviewer Questions
121
121
 
122
- Present each criticizer question with `ask_choice` (one call per question, in the configured language). Before each question, summarize the original criticism in at most three sentences and highlight the most important point. Do not revise the plan until every criticizer question has a recorded answer.
122
+ Merge the rounds' `Questions` sections into one deduped list and ask EVERY question with `ask_choice` (one batched `questions: [...]` form or one call per question, in the configured language, stable questionIds). Do not revise the plan until every reviewer question has a recorded answer.
123
123
 
124
124
  ### Round Lifecycle
125
125
 
126
- A refinement round is complete when all reviewer outputs have returned or all criticizer questions have answers. In the same turn: consolidate, accept or reject each finding on evidence (the user may override any disposition), revise to `PLAN_v(N+1).md` when accepted items require it (copy, edit only the new version, update the revision ledger and verifier checklist), then immediately ask the next merged accept/execute question. Never end a turn merely because a round completed.
126
+ A refinement round is complete when all reviewer lanes have returned; each lane's output carries findings (`F-###`) and up to five questions (`Q-1..Q-5`). In the same turn: consolidate, accept or reject each finding on evidence (the user may override any disposition), revise to `PLAN_v(N+1).md` when accepted items require it (copy, edit only the new version, update the revision ledger and verifier checklist), then immediately ask the next merged accept/execute question. Never end a turn merely because a round completed.
127
127
 
128
128
  ## Execution Handoff
129
129
 
130
- When the user picks `✓ Accept PLAN_vN and execute it now` in the merged question, mark the plan accepted and call the `execute_plan` tool (or the user runs `/plans-execute`). It re-confirms with the user, asks which runtime executes the plan (v0.6.0), then the extension enters execution mode:
130
+ When the user picks `✓ Accept PLAN_vN and execute it now` in the merged question, mark the plan accepted and call the `execute_plan` tool (or the user runs `/plans-execute`). It re-confirms with the user (never auto-completed), then the extension enters task-tree execution mode:
131
131
 
132
- - every agent turn is injected with the remaining verifier checklist and execution rules (layered simplest implementation, waiting for subprocess-backed verification with backoff 5s -> 10s -> 20s -> 40s -> 80s, then keep polling at 80s and restart at 5s for each new subprocess, no stopgaps, dependency and library discipline, minimum tests);
133
- - the runtime question offers ① the current session (recommended — injected checklist loop) and ② switching to another model; switching lists at least three `provider/model` switch targets (session-visible models plus the model registry, deduped, current excluded — `Other` free input when fewer exist) and runs ONE delegated executor subagent for the whole plan: native write tools, `PI_PLANS_EXECUTOR=1` + pinned `PI_PLANS_RUN_ID`, progress streamed into an `Executor` overlay while the parent parses `[DONE:VC-xxx]` markers from the child's full-text messages; Esc or `/plans-stop` kills the child and marks the run `stopped` (resumable under either runtime); under auto-approve or no-UI the question is skipped (current session, `[auto-approve]` recorded); implementation-review rounds keep the configured reviewer model;
134
- - execution-phase compaction is handled only when Pi core emits manual `/compact`, threshold, or overflow events; summaries are deterministic VCC-style summaries, include session-derived plan/current-I/checklist context, use smart tail keep and `keep:N`, and never call a model or request proactive current-I compaction;
132
+ - every agent turn is injected with the current wave's open tasks, the remaining task list, verification-check summary, and execution rules (wave order, `plans_update_task` reporting with status + evidence / skipReason, subprocess polling backoff 5s -> 10s -> 20s -> 40s -> 80s then keep polling at 80s, no stopgaps, dependency and library discipline, minimum tests);
133
+ - task progress flows exclusively through the `plans_update_task` tool: one call per task closing it as `complete` (with evidence) or `skipped` (with skipReason); closed statuses are immutable outside the audit-authorized rollback channel; subtasks close before their parent;
134
+ - the task dashboard tracks the whole tree live: a compact aboveEditor widget (current task ▸, progress bar, ✓/· counts, VC pass count, wave indicator, pause state, audit-failure line) and the Ctrl+Shift+T expanded tree view (✓/▸/~/· markers, current-task anchor, VC list with audit state, width-adaptive layout);
135
+ - a stall watchdog pauses execution after three consecutive settled rounds without any task-status change (genuine user input or `/plans-execute` resumes without losing progress);
136
+ - when every task reaches a terminal state, an independent read-only completion auditor verifies each check against the worktree: failed checks roll their covered tasks (children cascade, skipped tasks reopen) back to pending and inject the audit report; the audit re-runs as tasks re-close; after three failed rounds the run pauses for the user — under auto-approve/headless it terminates as `stopped` with the audit report so pipelines never hang; checks whose covered tasks are all skipped pass as skipped-pass; checks covering no task never enter the audit;
137
+ - execution-phase compaction is handled only when Pi core emits manual `/compact`, threshold, or overflow events; summaries are deterministic VCC-style summaries, include session-derived plan/current-task/checklist context, use smart tail keep and `keep:N`, and never call a model;
135
138
  - the read-only guard lifts: full write access returns;
136
- - the run status moves to `executing`, then `done` when the last `[DONE:VC-xxx]` marker lands;
139
+ - the run status moves to `executing`, then `done` when the completion audit passes;
137
140
  - `/plans-stop` stops execution; `/plans` shows progress.
138
141
 
139
142
  If the user declines, stay in planning (or stop, per their choice). Never start implementation without the approved handoff.
140
143
 
141
- ### Execution Goal-Wait
144
+ ### Continuation Between Turns
142
145
 
143
- In TUI/RPC, automatic goal-wait is evaluated only at `agent_settled`, after
146
+ In TUI/RPC, automatic continuation is evaluated only at `agent_settled`, after
144
147
  Pi has finished natural tool continuation, retries, and compaction. The
145
- extension rechecks that the same execution is active and incomplete, the
148
+ extension rechecks that the same execution is active with open tasks, the
146
149
  session is idle, and neither pending input nor compaction owns continuation.
147
150
  Each eligible settled cycle can send at most one hidden custom message with
148
- the current execution rules and remaining VC checklist. Tool `turn_end`
149
- events only update progress; they never prequeue goal-wait reminders.
150
-
151
- No-progress and literal `waiting for` counters advance only on eligible
152
- settled cycles. Real marker progress resets both; thresholds remain 3 and 6.
153
- User interruption and final model errors also pause continuation. Genuine
154
- interactive/RPC user input or `/plans-execute` can resume a paused active
155
- execution without losing verified VCs; extension input cannot unpause it.
156
- New-plan handoffs and the `execute_plan` tool still require explicit approval.
157
- Completion, stop, and session replacement invalidate the extension's wake
158
- identity without clearing user or other-extension queues. Print/JSON
159
- single-shot sessions keep VC tracking and completion but never auto-wake;
151
+ the current execution rules. Tool `turn_end` events only update usage; they
152
+ never prequeue continuation reminders. The stall watchdog counts settled
153
+ cycles without task-state change (threshold 3) and pauses instead of waking;
154
+ user interruption and final model errors pause too. Genuine interactive/RPC
155
+ user input or `/plans-execute` resumes a paused active execution without
156
+ losing task progress; extension input cannot unpause it. New-plan handoffs
157
+ and the `execute_plan` tool still require explicit approval. Print/JSON
158
+ single-shot sessions keep task tracking and completion but never auto-wake;
160
159
  use RPC for persistent headless execution.
161
160
 
162
- ### Post-Execution Continuation
163
-
164
- When execution completes in an interactive session, the completion message attaches a goal-running continuation block and triggers a new agent turn so the model can enter the implementation-review loop immediately. The interactive-only trigger keeps headless sessions silent (no unconsented subagent cost). The same behavior applies on both completion call sites (the normal `turn_end` completion and the `restoreFromSession` recovery path).
165
-
166
- The agent then asks two `ask_choice` questions (each single-question, `autoComplete: false`) to configure the implementation-review loop:
167
-
168
- 1. **Termination condition** (questionId `termination-condition`, recommended first: goal wait) — options: 1. goal wait: continue until no unpassed VCs remain (auto-continue each round) 2. until no high-severity finding (hard cap 5 rounds) 3. 1 round 4. 2 rounds 5. 3 rounds.
169
- 2. **Reviewer count** (questionId `impl-review-reviewer-count`, `allowOther: false`, pure digit labels `1`/`2`/`3`, recommended first): "How many concurrent reviewers should each implementation-review round use?" The recommended default follows the run's skill: `plan-big` / `plan-with-refs` → 3, others → 1.
170
-
171
- Both of these setup questions are AI-written options like any other: give each one a `description` of the form `✓ <advantage> / ✗ <drawback>` in the configured language (goal-wait buys thoroughness at the cost of a long run; more concurrent reviewers buy coverage at the cost of tokens), so the trade-off is visible before the user answers.
172
-
173
- Both answers persist TOGETHER in one `plans record-checkpoint` (`transition: "implementation-review-configured"`, `terminationCondition` + `reviewerCount`). Each refinement round calls `refine` with `role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>`; an omitted `reviewers` falls back to the run's configured `reviewerCount` from the checkpoint, so restarts and worktree migrations never silently revert 2/3 to 1. Each round accepts findings on evidence, applies fixes, re-runs relevant tests, and records progress. The hard cap is 5 rounds regardless of the chosen termination condition. Multi-reviewer rounds (2 or 3) consolidate like big-plan rounds: merge and dedupe findings into one `PLAN_vN_reviewer_comments.md`, keep each finding's source reviewer, severity, evidence, and disposition, and surface at most five high-priority findings. Crash recovery: if the process dies after one or both answers were recorded in `decisions.jsonl` but before the combined write, `/resume-plans` rebuilds the answered configuration from the ledger, asks only the missing question(s), and then performs the single combined write (the re-ask guard rejects duplicate configuration writes while the fields are already set).
174
- - Round audit trail: `decisions.jsonl`, `subagents.jsonl`, and `pi-plans-ameliorate` entries (one at goal start, then one per round) carry `currentRound` for post-hoc verification.
175
- - Headless sessions skip the prompt entirely; no `pi-plans-ameliorate` entry is appended.
176
-
177
161
  `refine` records each round and lane outcome durably (successful outputs are persisted to run-state files before the tool result returns) and accepts `resumeRoundId` to resume an interrupted round lane-by-lane: completed lanes are reused from their persisted outputs and never re-run; a round id is never reused across plan versions.
178
162
 
179
163
  ## Resuming (`/resume-plans`)
180
164
 
181
- After a restart or in a fresh session, `/resume-plans` (interactive only) restores the repository's working plan in the current session: the unfinished active run wins; otherwise a unique candidate resumes directly and multiple candidates get a chooser. It resumes unfinished planning (re-asks the pending question with the same `questionId`, never re-asks answered decisions), reviewing (resumes interrupted rounds via `refine resumeRoundId`, consolidates completed ones), execution (durable approval: unchanged plan digest keeps the authorization — a changed HEAD re-verifies old VCs first; legacy runs without checkpoints must re-approve), and the implementation-review loop (asks the termination condition and reviewer count only when they were never chosen — answered decisions are rebuilt from the ledger, never re-asked). While the loop is live, the pi-plans panel stays alive as a compact impl-review box (round count, reviewer count or "config pending", termination condition) and the status line mirrors it; once the checkpoint phase flips to `completed` the panel unregisters and the status line shows `(done)`. Linked worktrees share candidates; cross-worktree resumes confirm, copy artifacts without overwriting, reset approval and VC validity, and restart round counts. Record semantic boundaries with `plans record-checkpoint` (`plan-written`, `review-consolidated`, `implementation-review-configured`, `implementation-round-finished`, `completed` with evidence).
165
+ After a restart or in a fresh session, `/resume-plans` (interactive only) restores the repository's working plan in the current session: the unfinished active run wins; otherwise a unique candidate resumes directly and multiple candidates get a chooser. It resumes unfinished planning (re-asks the pending question with the same `questionId`, never re-asks answered decisions), reviewing (resumes interrupted rounds via `refine resumeRoundId`, consolidates completed ones), and execution (durable approval: unchanged plan digest keeps the authorization — a changed HEAD re-opens previously closed tasks for re-verification; an unverifiable approval HEAD behaves the same; legacy runs without checkpoints must re-approve; v0.6.0 delegated-executor orphans require a fresh handoff approval; a legacy `implementation-review` phase maps to done — its historical acceptance stands). Linked worktrees share candidates; cross-worktree resumes confirm, copy artifacts without overwriting, reset approval and VC validity, and restart round counts. Record semantic boundaries with `plans record-checkpoint` (`plan-written`, `review-consolidated`, `completed` with evidence).
182
166
 
183
- `refine` accepts a `target` parameter (`"plan"` default, `"implementation"` for the post-execution loop). The implementation brief anchors findings to the plan's goals and acceptance criteria, explicitly assesses delivery maturity (MVP-only vs. long-term refinement: stopgaps, missing tests, technical debt, production readiness), and tags out-of-scope improvements as low severity.
167
+ `refine` reviews the plan text (the v0.6.0 `target: "implementation"` post-execution loop is gone; the independent completion auditor now gates delivery).
184
168
 
185
169
  ## Red Flags
186
170
 
@@ -189,9 +173,9 @@ Stop and return to the workflow if any of these happen:
189
173
  - implementing before the approved execution handoff;
190
174
  - running `refine` without first asking the merged accept/execute question, or before the role gates pass;
191
175
  - ending a turn after a completed refinement round without asking the next merged accept/execute question;
192
- - storing planning settings outside the target workspace's `.git/pi_plans/` state directory;
176
+ - storing planning settings outside the target workspace's `.git/pi-plans/` state directory;
193
177
  - asking multiple planning questions in one message, or asking them outside `ask_choice`;
194
178
  - writing `PLAN_v1.md` before final scope confirmation;
195
179
  - accepting vague answers that contradict repo or reference evidence;
196
- - treating a reviewer or criticizer as authority instead of evidence;
180
+ - treating a reviewer as authority instead of evidence;
197
181
  - offering Auto-complete for execution, install, deploy, merge, push, or destructive cleanup approval.
@@ -3,79 +3,90 @@
3
3
  Status: draft | reviewed | accepted
4
4
  Plan version: N
5
5
  Artifact directory: `<artifact_root>/YYYY-MM-DD-topic/`
6
- State directory: `.git/pi_plans/runs/<run-id>/` (resolved git common dir)
6
+ State directory: `.git/pi-plans/runs/<run-id>/` (resolved git common dir)
7
7
  Language: `<BCP47 tag>`
8
8
 
9
9
  ## Original Request
10
10
 
11
- Summarize the user's request in one paragraph.
11
+ One paragraph summarizing the user's request.
12
12
 
13
- ## Goals
13
+ ## Tasks
14
14
 
15
- - `G-001`: Goal statement.
15
+ - `Task-1`: <title> — deps: <Task-ids, optional>; files: <paths, optional>; wave: <number, optional>
16
+ - `Task-2`: <title> — deps: Task-1; files: src/a.ts, src/b.ts; wave: 2
17
+ - `Task-2.1`: <subtask title> — deps: Task-1; files: src/a.ts
18
+ - `Task-3`: <title> — deps: Task-1, Task-2; files: src/c.ts; wave: 3
16
19
 
17
- ## Non-Goals
20
+ ### Execution Waves
18
21
 
19
- - `NG-001`: Explicitly excluded work.
22
+ - wave 1: Task-1 — <why these can run first / in parallel>
23
+ - wave 2: Task-2 — <files disjoint within the wave; deps satisfied by earlier waves>
24
+ - wave 3: Task-3 — <serial finish>
20
25
 
21
- ## Workspace State
26
+ ## Verification Checks
22
27
 
23
- - `STATE-001`: `.git/pi_plans/config.json` language, reviewer, and criticizer settings used for this run.
24
- - `STATE-002`: `.git/pi_plans/runs/<run-id>/run.json` and linked decision/subagent/ref ledgers.
25
-
26
- ## Repo Evidence
27
-
28
- - `E-REPO-001`: Path or command inspected, what it proves, and any uncertainty.
29
-
30
- ## External Evidence
31
-
32
- - `E-EXT-001`: URL or local ref path, what it supports, and date accessed.
33
-
34
- ## Resolved Decisions
35
-
36
- - `D-001`: Question, chosen answer, answer source (user | Auto-complete), and rationale.
37
-
38
- ## Requirements
39
-
40
- - `R-001`: Requirement tied to goals and decisions.
41
-
42
- ## Constraints
43
-
44
- - `C-001`: Compatibility, style, interface, performance, safety, or ownership constraint.
45
-
46
- ## Implementation Items
47
-
48
- - `I-001`: Work item with affected paths, dependencies, and expected code or doc changes.
49
-
50
- ## Acceptance Criteria
51
-
52
- - `AC-001`: Observable result tied to one or more requirements.
53
-
54
- ## Verification Plan
55
-
56
- - `V-001`: Command, manual check, screenshot, log review, or static inspection required after implementation.
57
-
58
- ## Verifier Checklist
59
-
60
- - [ ] `VC-001` covers `I-001`; pass condition: describe pass condition; evidence: describe expected evidence; metric: threshold or reason not quantified.
61
-
62
- ## Risks And Mitigations
63
-
64
- - `Risk-001`: Risk and mitigation.
65
-
66
- ## Refinement Settings
67
-
68
- - Reviewer mode: `delegated-subagent | current-session`; model selector: `inherit | <selector>`.
69
- - Criticizer mode: `delegated-subagent | current-session`; model selector: `inherit | <selector>`.
28
+ - [ ] `VC-001` covers `Task-1`; pass condition: <observable condition>; evidence: <expected evidence>; metric: <threshold or "not quantified">.
29
+ - [ ] `VC-002` covers `Task-2` and `Task-2.1`; pass condition: …; evidence: …; metric: ….
70
30
 
71
31
  ## Execution Handoff Notes
72
32
 
73
- State anything the executor should know, including order of work, files to avoid, and verification commands. The merged accept/execute question still requires explicit user approval (ask_choice with `autoComplete: false`, then the `execute_plan` tool) and must never be auto-completed. Once approved, the extension-managed execution loop injects the remaining checklist every turn and completes when every `[DONE:VC-xxx]` marker has landed — keep this section concise enough to serve as the executor's brief.
74
-
75
- ## Termination Recording (implementation review)
76
-
77
- When the post-execution implementation-review loop starts, the termination question is asked with `ask_choice` using `questionId: "termination-condition"` and persisted via `plans record-checkpoint` (`transition: "implementation-review-configured"`). Each disposed round records `implementation-round-finished`; the loop closes with `completed` plus evidence. These records make `/resume-plans` continue the loop with its original condition and round count.
33
+ Ordering, files to avoid, verification commands, and anything the executor must know. The handoff still requires explicit user approval (`ask_choice` with `autoComplete: false`, then the `execute_plan` tool) and is never auto-completed. Once approved, execution mode tracks every task through the `plans_update_task` tool (status + evidence); when all tasks are terminal, the independent completion auditor verifies each check above before the run completes.
78
34
 
79
35
  ## Revision Ledger
80
36
 
81
- - `PLAN_v1`: Initial plan from resolved questions and evidence.
37
+ - `PLAN_v1`: <one line per revision: what changed and why>.
38
+
39
+ ---
40
+
41
+ ## Format specification (normative)
42
+
43
+ The plan body is exactly two sections plus the metadata header shown above:
44
+ `## Original Request` (one paragraph), `## Tasks`, and `## Verification Checks`.
45
+ Everything else the workflow needs lives in the run state (decisions ledger,
46
+ review rounds, refs), not in the plan file. `## Execution Handoff Notes` and
47
+ `## Revision Ledger` are the two permitted auxiliary sections.
48
+
49
+ ### Tasks microsyntax
50
+
51
+ - Top-level task line: ``- `Task-N`: <title> — deps: <ids>; files: <paths>; wave: <n>``.
52
+ The separator between title and metadata is an em dash `—` (tolerated: `——`, `--`, `–`).
53
+ With no separator the whole body is the title (no fields).
54
+ - Fields are separated by `;` (tolerated `;`); multi-values by `,`
55
+ (tolerated `,` `、`). All three fields are optional; missing `wave` values
56
+ derive from deps (1 + max(dep wave)), and the `### Execution Waves`
57
+ subsection wins over inline `wave:` on conflict (linted).
58
+ - Subtasks: one indented bullet level, ``- `Task-N.M`: <title> — …``. Only one
59
+ nesting level is valid; deeper ids lint as drift. Subtasks may carry
60
+ `deps:`/`files:`; the wave is inherited from the parent (inline `wave:` on a
61
+ subtask is ignored with a lint notice).
62
+ - `### Execution Waves` rows: ``- wave <n>: Task-1, Task-2 — <rationale>``.
63
+ The wave table is the authoritative parallel-execution order: within one
64
+ wave, top-level tasks' file sets must be disjoint (children roll up to their
65
+ parent), and every `deps:` target must sit in an earlier wave.
66
+
67
+ ### Verification Checks microsyntax
68
+
69
+ - Row: ``- [ ] `VC-###` covers `Task-2` and `Task-3`; pass condition: …; evidence: …; metric: …``.
70
+ - `covers` accepts multiple targets and `Task-N.M` subtask ids; the clause ends
71
+ at the first `;`. Checks covering zero tasks never enter the completion
72
+ audit; a check whose covered tasks are ALL skipped passes as skipped-pass.
73
+
74
+ ### Lint and compatibility
75
+
76
+ - Planning-time lint (`lintPlanIntoNotices`): zero parsed tasks under an
77
+ existing `## Tasks` header, over-deep subtasks, non-consecutive top-level
78
+ numbering, unknown dep/coverage targets, same-wave file overlaps,
79
+ deps inside the same-or-later wave, inline-vs-subsection wave conflicts.
80
+ Lint notices are advisory while planning and hard-rejected at the execution gate.
81
+ - Legacy compatibility: plans without `## Tasks` fall back to parsing
82
+ `## Implementation Items` (`I-001` → `Task-1`, `covers \`I-001\`` normalizes
83
+ the same way); checklist-only pre-0.5 artifacts synthesize one serial task
84
+ per check. The execution gate surfaces an upgrade notice on fallback.
85
+
86
+ ### Refinement
87
+
88
+ One Reviewer role: each round returns findings (`F-###`) and up to five
89
+ questions (`Q-1..Q-5`). Default sequences: plan-big → one round of three
90
+ concurrent reviewers (questions included); plan-normal / plan-small → one
91
+ reviewer round. The main agent asks every question with `ask_choice` and
92
+ records the answers before revising the plan.