pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
@@ -1,18 +1,20 @@
1
1
  # State And Config
2
2
 
3
- pi-plans stores all planning preferences and run state in the target workspace's git directory as `<git-common-dir>/pi_plans/` — in an ordinary repository this is simply `.git/pi_plans/` — resolving the git common dir with `git rev-parse --git-common-dir` from the workspace. Because the state lives inside the git dir, git never tracks it and no `.gitignore` entries are needed. The target workspace is the current working directory unless the user explicitly names another repository.
3
+ pi-plans stores planning preferences and run state in the target workspace's git directory as `<git-common-dir>/pi-plans/` — in an ordinary repository this is simply `.git/pi-plans/` — resolving the git common dir with `git rev-parse --git-common-dir` from the workspace. Because the state lives inside the git dir, git never tracks it and no `.gitignore` entries are needed. The target workspace is the current working directory unless the user explicitly names another repository.
4
4
 
5
- Do not store pi-plans preferences in Pi's own settings (`~/.pi/agent/settings.json`); pi-plans uses `.git/pi_plans/config.json` for its state.
5
+ The one workspace-independent piece is the **reviewer role** (v0.7.0): it lives in the global config `~/.pi/pi-plans/config.json` (override the directory with `PI_PLANS_GLOBAL_DIR`), confirmed once and shared across every workspace. It is a standalone pi-plans file, never Pi's own settings.
6
+
7
+ Do not store pi-plans preferences in Pi's own settings (`~/.pi/agent/settings.json`); pi-plans uses `.git/pi-plans/config.json` for workspace state and `~/.pi/pi-plans/config.json` for the reviewer role.
6
8
 
7
9
  ## State Root Resolution
8
10
 
9
11
  - Git runs with `GIT_DIR`, `GIT_COMMON_DIR`, and `GIT_WORK_TREE` scrubbed from the environment, so leaked env vars cannot misdirect state into an unrelated repository. Relative results (`.git`, `../.git`) resolve against the workdir.
10
12
  - Granularity is **per enclosing repository**: running from a subdirectory uses the enclosing repo's git dir (a one-line notice names that repo). Linked worktrees share one common dir; run directories are unique, and since v0.6.0 the run registry is derived from `runs/` (no shared pointer to race). The legacy `active.json` is deprecated: reads fall back to it only when no `runs/` entries exist (pre-0.6.0 migration).
11
- - State does not travel with clones: a fresh clone starts with empty state while committed `./docs/pi-plans/` artifacts persist in the repository.
13
+ - State does not travel with clones: a fresh clone starts with empty state, and the default plan artifacts live in the git dir with it. Point `artifact_root` at `./docs/pi-plans` when you want the plans committed and public instead.
12
14
 
13
15
  ## Auto Git Init
14
16
 
15
- When a mutating state action (`init`, `set-language`, `set-role`, `start-run`, record-*) runs in a workdir that is not a git repository, the helper auto-runs `git init` there (with a one-line notice) and then creates the state dir. It never creates commits. Auto-init runs only when ALL of the following hold:
17
+ When a mutating state action (`init`, `set-language`, `set-artifact-root`, `set-refs-root`, `set-graph-enabled`, `start-run`, record-*) runs in a workdir that is not a git repository, the helper auto-runs `git init` there (with a one-line notice) and then creates the state dir. It never creates commits. `set-role` is deliberately NOT in this list: it writes only the global config and never triggers auto-init or any workspace write. Auto-init runs only when ALL of the following hold:
16
18
 
17
19
  - the workdir has no `.git` entry (a pre-existing `.git` file or directory that git cannot resolve is a fatal error, never a silent reinit);
18
20
  - the workdir is not inside any git work tree (a subdirectory of a repo uses the enclosing repo instead);
@@ -23,12 +25,12 @@ Bare repositories are refused with a clear error. A missing `git` executable is
23
25
  ## Directory Layout
24
26
 
25
27
  ```text
26
- <git-common-dir>/pi_plans/
28
+ <git-common-dir>/pi-plans/
27
29
  config.json
28
30
  pi-vcc-config.json
29
31
  active.json # deprecated (v0.6.0): legacy pointer, read only when runs/ is empty
30
32
  runs/ # the run registry derives from runs/<run-id>/run.json
31
- <run-id>/ # note: the refs root is a sibling — .git/pi-plans/refs (hyphenated), not under pi_plans/
33
+ <run-id>/ # note: the refs root is a sibling — .git/pi-plans/refs (hyphenated), not under pi-plans/
32
34
  run.json
33
35
  decisions.jsonl
34
36
  subagents.jsonl
@@ -41,25 +43,13 @@ Bare repositories are refused with a clear error. A missing `git` executable is
41
43
 
42
44
  ## Config Schema
43
45
 
44
- The default config is:
46
+ The default workspace config is (no `reviewer` key — the role lives globally since v0.7.0):
45
47
 
46
48
  ```json
47
49
  {
48
50
  "schema": 1,
49
51
  "language": { "tag": null, "source": "unset", "updated_at": null },
50
- "reviewer": {
51
- "mode": "delegated-subagent",
52
- "model_selector": null,
53
- "name_prefix": "pi-plans-reviewer",
54
- "confirmed_at": null
55
- },
56
- "criticizer": {
57
- "mode": "delegated-subagent",
58
- "model_selector": null,
59
- "name_prefix": "pi-plans-criticizer",
60
- "confirmed_at": null
61
- },
62
- "artifact_root": "./docs/pi-plans",
52
+ "artifact_root": "./.git/pi-plans/plans",
63
53
  "artifact_root_source": "unset",
64
54
  "artifact_root_updated_at": null,
65
55
  "refs_root": null,
@@ -72,18 +62,42 @@ Rules:
72
62
 
73
63
  - `schema` must be `1`.
74
64
  - `language.tag` is a BCP47-style tag such as `zh-Hans`, `en`, or `zh-Hant`, or `null` before selection; `language.source` is `user`, `auto`, or `unset`.
75
- - `reviewer.mode` and `criticizer.mode` are `delegated-subagent` or `current-session`.
76
- - `model_selector` is `null` to inherit the dispatching session's model, or an exact `provider/model` selector matching Pi's model registry.
77
- - `confirmed_at` is `null` until the user has confirmed the role's model at first use; see below.
65
+ - A legacy workspace `reviewer` block (pre-0.7.0, and the removed v0.6.0 `criticizer` key) is read-tolerated: the first mutating state call seeds the global config from intent blocks and the next workspace write strips the key. Read-only paths resolve the effective reviewer in memory (global first, legacy block second) and never write.
78
66
  - `artifact_root` is relative to the target workspace unless absolute.
79
67
  - `artifact_root_source` is `user`, `auto`, or `unset`.
80
68
  - `artifact_root_updated_at` is the selection timestamp or `null` before confirmation.
81
69
  - `refs_root` is where plan-with-refs downloads references, relative to the target workspace unless absolute, or `null` before selection; `refs_root_source` is `user`, `auto`, or `unset`; `refs_root_updated_at` is the selection timestamp or `null`.
82
- - There is intentionally no `effort` field: subagents inherit the dispatching session's model and thinking level unless an exact selector is stored. The real lever is the main session's thinking level at refine time.
70
+
71
+ ## Global Reviewer Config (v0.7.0)
72
+
73
+ The reviewer role is stored in `~/.pi/pi-plans/config.json` (`PI_PLANS_GLOBAL_DIR` overrides the directory; tests, CI, and bench containers rely on it):
74
+
75
+ ```json
76
+ {
77
+ "schema": 1,
78
+ "reviewer": {
79
+ "mode": "delegated-subagent",
80
+ "model_selector": "devin/claude-sonnet-5.5",
81
+ "thinking_level": null,
82
+ "name_prefix": "pi-plans-reviewer",
83
+ "confirmed_at": "2026-09-30T07:54:12Z"
84
+ }
85
+ }
86
+ ```
87
+
88
+ Rules:
89
+
90
+ - `mode` is `delegated-subagent` or `current-session`.
91
+ - `model_selector` is an exact `provider/model` selector. There is no inherit entry point anymore: after first-use confirmation the delegated reviewer always runs a concrete model (`modelSelector: "inherit"` in `set-role` is a full reset — it clears the selector AND `confirmed_at`).
92
+ - `thinking_level` is `null` (default — spawn WITHOUT `--thinking`, letting the child pi resolve its own chain: per-model settings → `defaultThinkingLevel` → `medium`, then model clamping) or an explicit level (`off | minimal | low | medium | high | xhigh | max`; the domain comes from the chosen model's `thinkingLevelMap` via pi-ai's `getSupportedThinkingLevels`). `"off"` is a real level and deliberately distinct from `null`. Changing `modelSelector` without passing `thinkingLevel` resets the level.
93
+ - `confirmed_at` is stamped only by a real confirmation. `reviewerReady(role)` = current-session, or delegated with `confirmed_at` set AND a concrete `model_selector` — a confirmed null selector can never pass (the old confirmed-inherit state is unreachable).
94
+ - A corrupt or wrong-schema global file yields defaults plus a notice and is NEVER clobbered by reads.
95
+ - Migration (Q-1=A): the first mutating pi-plans call in a workspace with a legacy intent block (`confirmed_at` set, an explicit selector, or a non-default mode) seeds the global file once — first touched workspace wins; other workspaces get a one-time "ignored" notice. A confirmed-inherit block seeds with the selector null and NO confirmation, so the next `refine` re-asks once via the native panel. Scaffold-only blocks are dropped silently.
96
+ - The completion auditor's spawns are NOT governed by this role: it runs its own model; `subagents.jsonl` records the reviewer/ref-analyst thinking level actually passed (`thinking_level` field, older entries have none).
83
97
 
84
98
  ## VCC Compact Config
85
99
 
86
- `pi-vcc-config.json` is scaffolded under the resolved `<git-common-dir>/pi_plans/` state root when an active planning or execution compaction hook first needs it. It is independent from `config.json` so planning preferences, run state, and compact policy can evolve separately.
100
+ `pi-vcc-config.json` is scaffolded under the resolved `<git-common-dir>/pi-plans/` state root when an active planning or execution compaction hook first needs it. It is independent from `config.json` so planning preferences, run state, and compact policy can evolve separately.
87
101
 
88
102
  Default values:
89
103
 
@@ -117,7 +131,7 @@ Before the first product planning question, check the persisted config (`plans`
117
131
  4. `Other` — user provides a BCP47 tag.
118
132
  5. `Auto-complete` — select the recommended language.
119
133
 
120
- Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selected language for visible questions, choices, review summaries, criticizer questions, and Markdown artifacts. Keep IDs, file paths, command names, JSON keys, and protocol labels stable in English.
134
+ Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selected language for visible questions, choices, review summaries, reviewer questions, and Markdown artifacts. Keep IDs, file paths, command names, JSON keys, and protocol labels stable in English.
121
135
 
122
136
  ## Code Graph Enabled
123
137
 
@@ -125,13 +139,13 @@ Persist with `plans` (`set-language`, `languageSource: "user"`). Use the selecte
125
139
 
126
140
  ## `/config-pi-plans`
127
141
 
128
- `/config-pi-plans` is an interactive workspace configuration wizard. It re-asks the workspace language, planning docs root, refs root, code graph toggle, reviewer mode/model, and criticizer mode/model, then writes the chosen defaults back to `.git/pi_plans/config.json`. When code graph is enabled, the extension also overrides built-in `read`/`write`/`edit` for indexed source files so graph-backed source reads and DB-first edits happen automatically. Model pickers can reuse the current session model, any available selector surfaced by `ctx.scopedModels` or the model registry, or a manually entered exact `provider/model` string. If a run is already active, only the workspace defaults change; the active run's `artifact_dir` and `language_tag` stay unchanged.
142
+ `/config-pi-plans` is an interactive workspace configuration wizard. It re-asks the workspace language, planning docs root, refs root, and code graph toggle (written to `.git/pi-plans/config.json`), plus the reviewer mode and model. The reviewer steps live in the GLOBAL config: the mode switch persists immediately, and the model step shows a keep/change entry menu — `Keep current (provider/model · level)` or `Choose model & thinking level…`. When the mode is `current-session` the model step is skipped entirely (an existing selector is never cleared). Choosing opens the native model panel + effort panel in TUI, or model/effort menus otherwise; Esc keeps the current role and the wizard CONTINUES instead of discarding earlier answers. A failed global write is reported explicitly while workspace settings still save. When code graph is enabled, the extension also overrides built-in `read`/`write`/`edit` for indexed source files so graph-backed source reads and DB-first edits happen automatically. If a run is already active, only the workspace defaults change; the active run's `artifact_dir` and `language_tag` stay unchanged.
129
143
 
130
144
 
131
145
  Before the first product planning question, check the persisted config again. If `artifact_root_source` is missing or `unset`, ask exactly one `ask_choice` question:
132
146
 
133
- 1. `./docs/pi-plans` — recommended; planning docs live in the repository and are public.
134
- 2. `./.git/pi_plans/plans` — private to the repository; not published.
147
+ 1. `./.git/pi-plans/plans` — recommended; the default. Planning docs stay private to the repository and are never tracked.
148
+ 2. `./docs/pi-plans` — planning docs live in the working tree, are public, and can be committed with the repository.
135
149
  3. `Other` — user provides a custom path.
136
150
  4. `Auto-complete` — select the recommended path.
137
151
 
@@ -147,27 +161,29 @@ Before downloading any reference in a plan-with-refs flow, check the persisted c
147
161
  Persist with `plans` (`set-refs-root`, `refsRoot: <selected path>`, `refsRootSource: "user"` or `"auto"`). Download references under this root. This question does not count against the planning-question limit.
148
162
 
149
163
 
150
- Before running a `refine` round, read the role setting from the persisted config.
164
+ Before running a `refine` round, read the role setting from the persisted global config.
151
165
 
152
- If the role's `mode` is missing or invalid, ask exactly one `ask_choice` question and persist:
166
+ If the role's `mode` is missing or invalid, ask exactly one `ask_choice` question and persist (the mode question stays agent-mediated):
153
167
 
154
168
  1. `Delegated subagent` — recommended; read-only `pi` subprocess with isolated context.
155
169
  2. `Current session` — run the read-only pass in the current foreground session.
156
170
  3. `Other` / 4. `Auto-complete` — select the recommended delegated subagent.
157
171
 
158
- Independently, each role's **model** is confirmed once, at that role's first actual use: when `confirmed_at` is `null` and a `refine` round is about to run, ask exactly one `ask_choice` question:
172
+ The **model + thinking level** are confirmed once, at first actual use, through NATIVE panels — not ask_choice:
173
+
174
+ - **TUI**: the gate itself pops a `/model`-style searchable model panel (pi's `ModelSelectorComponent`; the runtime adapter maps the public model registry facade — `getAvailable`/`find`/`getError`/`refresh` — onto the four runtime methods, with the private `.runtime` field as a secondary attempt and menus as the construction fallback), then a `/thinking`-style effort panel whose first row is `Default (no --thinking flag: the child pi resolves per-model settings → defaultThinkingLevel → medium)` followed by the chosen model's `thinkingLevelMap` levels. Both panels must complete; the result persists to the global config (`modelSelector` + `thinkingLevel`, `confirmed: true`) only then, and the same invocation continues with the returned values.
175
+ - **hasUI non-TUI (RPC/ACP)**: native `ctx.ui.select` menus over the same data (model list, then effort list).
176
+ - **UI-less (print/json/bench)**: the gate returns text guidance embedding the available selectors and the exact `set-role` call; persist that way. Automation may also pre-write `~/.pi/pi-plans/config.json` directly.
159
177
 
160
- 1. `Inherit the main agent's model` — recommended; stores `model_selector: null`.
161
- 2. `Choose a model` — pick from the models available in this Pi install (check `/model` or `ctx.scopedModels`); persist the exact `provider/model` selector; do not invent model names.
162
- 3. `Other` / 4. `Auto-complete` — select inherit.
178
+ Pressing Esc on either panel CANCELS the whole gate: nothing is persisted and the tool returns a dedicated error (`details.cancelled`) instructing the agent NOT to re-ask via ask_choice and NOT to retry unless the user asks — suggest `/config-pi-plans` instead. Cheap validations (plan path, refs) run BEFORE any panel so a typo never walks the user through panels.
163
179
 
164
- Persist with `plans` (`set-role`, `confirmed: true`, `modelSelector: <selector or "inherit">`). `confirmed_at` is set only by this confirmation flow; a mode-only edit never forges or discards a confirmation, and a confirmed inherit (`model_selector: null` plus a stamp) is distinguishable from never-confirmed.
180
+ The confirmation applies only to `delegated-subagent` mode (current-session runs in the main session and needs no model). If a stored selector is missing from the registry at spawn time, TUI re-opens the panel; other modes return an error naming the selector — reset with `set-role` (`resetConfirmation: true`) and re-confirm.
165
181
 
166
- If a spawn later fails because the stored selector is unavailable, reset the marker (`set-role`, `resetConfirmation: true`) and re-ask the confirmation question.
182
+ `set-role` invariants: `confirmed: true` requires a concrete `provider/model` selector in delegated mode; `modelSelector: "inherit"` resets BOTH the selector and `confirmed_at`; `thinkingLevel: "default"` stores `null`; changing `modelSelector` without `thinkingLevel` resets the level.
167
183
 
168
184
  ## Subagent Spawning
169
185
 
170
- When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` subprocess (`--mode json -p --no-session --tools read,grep,find,ls`) whose system prompt comes from `agents/reviewer.md` or `agents/criticizer.md`. In TUI mode, delegated runs also show a standalone `Reviewer` or `Criticizer` overlay with live lane/tool status; the child is awaited and the overlay is closed before the tool result returns. The subagent:
186
+ When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` subprocess (`--mode json -p --no-session --tools read,grep,find,ls`, plus `--model <provider/model>` and — only when `thinking_level` is set — `--thinking <level>`) whose system prompt comes from `agents/reviewer.md`. In TUI mode, delegated runs also show a standalone `Reviewer` overlay with live lane/tool status (header label `provider/model:level`); the child is awaited and the overlay is closed before the tool result returns. The subagent:
171
187
 
172
188
  - performs read-only analysis and never edits files;
173
189
  - receives the full plan text and a review/criticism brief;
@@ -175,19 +191,19 @@ When `mode` is `delegated-subagent`, the `refine` tool spawns a read-only `pi` s
175
191
 
176
192
  The main agent consolidates the results, records dispositions, revises the plan, and asks the next merged accept/execute question — all in the same turn.
177
193
 
178
- The `analyze_refs` tool (plan-with-refs) uses the same spawning machinery with the **reviewer** role's gates (`mode` must be `delegated-subagent`; a confirmed `current-session` reviewer is refused with guidance to switch, since analysis is spawn-only) and the reviewer's model selector. Each downloaded reference gets one independent read-only subagent whose system prompt comes from `agents/ref-analyst.md` and whose working directory is that reference's own directory; lanes never get `code_graph`. Lanes run in sequential batches of at most 3 under a standalone overlay titled `Refs`; each batch's controller opens and closes exactly like a single refine round. Successful spawns are recorded best-effort in `subagents.jsonl` with role `ref-analyst` (skipped when no active run exists, e.g. adhoc calls). The structured per-reference sections come back as the tool result; the main agent owns `REF_ANALYSIS.md` and fills `coverage`/`gaps` in `refs.jsonl` via `plans` (`record-ref`).
194
+ The `analyze_refs` tool (plan-with-refs) uses the same spawning machinery with the **reviewer** role's model confirmation from the global config — but NOT its mode (v0.7.0, Q-4): analysis is spawn-only by nature, so a `current-session` reviewer still gets spawned ref-analyst lanes (a one-time notice says the mode is ignored and unchanged) while the confirmed concrete model remains required. Each downloaded reference gets one independent read-only subagent whose system prompt comes from `agents/ref-analyst.md` and whose working directory is that reference's own directory; lanes never get `code_graph`. Lanes run in sequential batches of at most 3 under a standalone overlay titled `Refs`; each batch's controller opens and closes exactly like a single refine round. Successful spawns are recorded best-effort in `subagents.jsonl` with role `ref-analyst` and the thinking level actually passed (skipped when no active run exists, e.g. adhoc calls). The structured per-reference sections come back as the tool result; the main agent owns `REF_ANALYSIS.md` and fills `coverage`/`gaps` in `refs.jsonl` via `plans` (`record-ref`).
179
195
 
180
196
  ## Run State
181
197
 
182
- One run directory per planning request: `<git-common-dir>/pi_plans/runs/<YYYYMMDDTHHMMSSZ-topic>/` (second-precision; `-2`, `-3` suffixes on collision).
198
+ One run directory per planning request: `<git-common-dir>/pi-plans/runs/<YYYYMMDDTHHMMSSZ-topic>/` (second-precision; `-2`, `-3` suffixes on collision).
183
199
 
184
200
  `run.json` includes: run ID; skill name; original request; target workspace; artifact directory; language tag; status (`planning` → `accepted` → `executing` → `done`, with `stopped`/`abandoned` as exits); timestamps.
185
201
 
186
- `decisions.jsonl` is appended automatically by `ask_choice` (question, options, answer, answer source). `subagents.jsonl` records reviewer/criticizer/ref-analyst spawns. `refs.jsonl` records reference metadata via `plans` (`record-ref`); its `kind` field is `project` (repos), `paper` (arXiv etc.), `article` (blog posts), or `docs` (documentation sites).
202
+ `decisions.jsonl` is appended automatically by `ask_choice` (question, options, answer, answer source). `subagents.jsonl` records reviewer/ref-analyst spawns (legacy 0.6.0 entries with `criticizer` remain readable). `refs.jsonl` records reference metadata via `plans` (`record-ref`); its `kind` field is `project` (repos), `paper` (arXiv etc.), `article` (blog posts), or `docs` (documentation sites).
187
203
 
188
204
  ## Workflow Checkpoints (`/resume-plans`)
189
205
 
190
- Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | implementation-review | completed`), `nextAction`, the exact plan identity (path + version + SHA-256), pending/answered questions (stable `questionId`), review rounds with per-lane status and result-file references, execution approval evidence (plan digest, worktree, `git rev-parse HEAD` at approval, verified VC/I set, usage), the implementation-review termination condition and completed-round count, and ownership metadata. Full review outputs live in separate `reviews/` files; the checkpoint keeps only validated references.
206
+ Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | completed`; the legacy 0.6.0 `implementation-review` phase is read-tolerated and maps to done), `nextAction`, the exact plan identity (path + version + SHA-256), pending/answered questions (stable `questionId`), review rounds with per-lane status and result-file references, execution approval evidence (plan digest, worktree, `git rev-parse HEAD` at approval, task progress map, audit rounds, verified VC set, usage), and ownership metadata. Full review outputs live in separate `reviews/` files; the checkpoint keeps only validated references.
191
207
 
192
208
  Rules:
193
209
 
@@ -200,4 +216,4 @@ Rules:
200
216
 
201
217
  ## Run Ownership
202
218
 
203
- A run may be held by at most one live owner (`owner.json`: host, pid, process start time via `ps -o lstart=`, session id, random process token, generation). Acquisition is an atomic exclusive create; takeovers require proof the previous owner is dead (process gone, or pid alive with a different start time — PID reuse). Foreign hosts, corrupt records, and unverifiable liveness are conservatively refused; `/resume-plans` never queues or interrupts. Sessions bind to the run they start/execute/resume (restored from `pi-plans-run-start` entries on the current branch), and attribution (tools, write guard, autocomplete, execution bookkeeping, code-graph apply gate) prefers the binding, falling back to the registry's newest non-terminal run (v0.6.0 — the shared `active.json` pointer is deprecated). Delegated executor children pin their run via `PI_PLANS_RUN_ID` and are exempt from the planning write guard (`PI_PLANS_EXECUTOR=1`).
219
+ A run may be held by at most one live owner (`owner.json`: host, pid, process start time via `ps -o lstart=`, session id, random process token, generation). Acquisition is an atomic exclusive create; takeovers require proof the previous owner is dead (process gone, or pid alive with a different start time — PID reuse). Foreign hosts, corrupt records, and unverifiable liveness are conservatively refused; `/resume-plans` never queues or interrupts. Sessions bind to the run they start/execute/resume (restored from `pi-plans-run-start` entries on the current branch), and attribution (tools, write guard, autocomplete, execution bookkeeping, code-graph apply gate) prefers the binding, falling back to the registry's newest non-terminal run (v0.6.0 — the shared `active.json` pointer is deprecated). The v0.6.0 delegated-executor env pins (`PI_PLANS_RUN_ID` / `PI_PLANS_EXECUTOR`) are gone; read-only subagent children carry `PI_PLANS_REFINER=1`.
@@ -25,7 +25,7 @@ Usage with harbor::
25
25
 
26
26
  Fairness (D-010/D-020): both arms share one container image; the only
27
27
  difference is the agent-level configuration above. Seeded evaluation state
28
- lives under ``.git/pi_plans/`` with the artifact root pointed OUTSIDE the
28
+ lives under ``.git/pi-plans/`` with the artifact root pointed OUTSIDE the
29
29
  graded tree (``/tmp/pi-plans-bench``); the pre-registered pre-oracle
30
30
  snapshot-diff restore was NOT implemented in this run — recorded as a
31
31
  limitation (TB oracles read /app artifacts only, so scoring impact is
@@ -63,27 +63,33 @@ TREATMENT_SYSTEM_PROMPT = (
63
63
  "it step by step with the verifier checklist."
64
64
  )
65
65
 
66
- # D-015 seeded config: deterministic, no first-use Q&A rounds, flash roles,
67
- # graph off, artifact root outside the graded workspace.
66
+ # D-015 seeded configs: deterministic, no first-use Q&A rounds, graph off,
67
+ # artifact root outside the graded workspace. Since v0.7.0 the reviewer role
68
+ # lives in the GLOBAL config (~/.pi/pi-plans/config.json inside the
69
+ # container) with a CONCRETE provider/model selector (inherit was removed);
70
+ # the workspace config carries no reviewer/criticizer keys at all.
68
71
  SEEDED_CONFIG = {
69
72
  "schema": 1,
70
73
  "artifact_root": "/tmp/pi-plans-bench/docs",
71
74
  "artifact_root_source": "user",
72
75
  "language": {"tag": "en", "source": "user"},
73
- "reviewer": {
74
- "mode": "delegated-subagent",
75
- "model_selector": None, # None => inherit the main agent's model (flash)
76
- "confirmed_at": "1970-01-01T00:00:00Z",
77
- },
78
- "criticizer": {
79
- "mode": "delegated-subagent",
80
- "model_selector": None,
81
- "confirmed_at": "1970-01-01T00:00:00Z",
82
- },
83
76
  "graph_enabled": False,
84
77
  }
85
78
 
86
79
 
80
+ def _seeded_global_config(provider: str, model_id: str) -> dict:
81
+ return {
82
+ "schema": 1,
83
+ "reviewer": {
84
+ "mode": "delegated-subagent",
85
+ "model_selector": f"{provider}/{model_id}",
86
+ "thinking_level": None, # default: child pi resolves its own chain
87
+ "name_prefix": "pi-plans-reviewer",
88
+ "confirmed_at": "1970-01-01T00:00:00Z",
89
+ },
90
+ }
91
+
92
+
87
93
  def _read_text(name: str) -> str:
88
94
  return (ADAPTER_DIR / name).read_text(encoding="utf-8")
89
95
 
@@ -199,20 +205,25 @@ class PiPlansBench(Pi):
199
205
  filename=".bench-system-prompt.md",
200
206
  )
201
207
 
202
- async def _seed_pi_plans_config(self, environment: BaseEnvironment) -> None:
203
- """Write the deterministic pi-plans config into the trial workdir (D-015).
208
+ async def _seed_pi_plans_config(self, environment: BaseEnvironment, provider: str, model_id: str) -> None:
209
+ """Write the deterministic pi-plans configs into the trial container (D-015).
204
210
 
205
- The config lives under ``.git/pi_plans/`` (diff-allowlist path, removed
206
- before oracle scoring) with the artifact root pointed at
207
- ``/tmp/pi-plans-bench`` so plan artifacts never land in the graded tree.
211
+ The workspace config lives under ``.git/pi-plans/`` (diff-allowlist
212
+ path, removed before oracle scoring) with the artifact root pointed at
213
+ ``/tmp/pi-plans-bench`` so plan artifacts never land in the graded
214
+ tree. The reviewer role is seeded into the GLOBAL config under the
215
+ container's ``~/.pi/pi-plans/`` with the exact driver model (v0.7.0:
216
+ no inherit, no criticizer), so refine gates pass without any panel.
208
217
  """
209
218
  config = json.dumps(SEEDED_CONFIG, indent=2)
219
+ global_config = json.dumps(_seeded_global_config(provider, model_id), indent=2)
210
220
  await self.exec_as_agent(
211
221
  environment,
212
222
  command=(
213
223
  "set -euo pipefail; "
214
- "mkdir -p .git/pi_plans " f"{BENCH_CONFIG_DIR} && "
215
- f"printf {shlex.quote(config)} > .git/pi_plans/config.json"
224
+ "mkdir -p .git/pi-plans ~/.pi/pi-plans " f"{BENCH_CONFIG_DIR} && "
225
+ f"printf {shlex.quote(config)} > .git/pi-plans/config.json && "
226
+ f"printf {shlex.quote(global_config)} > ~/.pi/pi-plans/config.json"
216
227
  ),
217
228
  )
218
229
 
@@ -224,7 +235,7 @@ class PiPlansBench(Pi):
224
235
  provider, model_id = self.model_name.split("/", 1)
225
236
 
226
237
  if self.arm == "treatment":
227
- await self._seed_pi_plans_config(environment)
238
+ await self._seed_pi_plans_config(environment, provider, model_id)
228
239
 
229
240
  # printf interprets backslash escapes; ship sources base64-encoded so
230
241
  # the driver/task bytes survive verbatim.
@@ -291,7 +302,7 @@ class PiPlansBench(Pi):
291
302
  if subagent_usage is not None:
292
303
  metadata["pi_plans_subagent_usage"] = subagent_usage
293
304
  # Fold child tokens/cost into the top-level accounting so treatment
294
- # totals are comparable (the extension's reviewer/criticizer calls).
305
+ # totals are comparable (the extension's reviewer/ref-analyst calls).
295
306
  totals = subagent_usage.get("totals") or {}
296
307
  context.n_input_tokens = (context.n_input_tokens or 0) + int(totals.get("input", 0))
297
308
  context.n_output_tokens = (context.n_output_tokens or 0) + int(totals.get("output", 0))
@@ -184,11 +184,11 @@ function snapshotPlanArtifacts(dest) {
184
184
  if (!dest) return;
185
185
  // Plan artifact locations: the seeded artifact root (/tmp/pi-plans-bench/docs)
186
186
  // and pi-plans' own fallbacks — the repo-local docs/pi-plans and the state
187
- // root .git/pi_plans/docs (where run plans actually land).
187
+ // root .git/pi-plans/docs (where run plans actually land).
188
188
  const roots = [
189
189
  "/tmp/pi-plans-bench/docs",
190
190
  path.join(process.cwd(), "docs", "pi-plans"),
191
- path.join(process.cwd(), ".git", "pi_plans", "docs"),
191
+ path.join(process.cwd(), ".git", "pi-plans", "docs"),
192
192
  ];
193
193
  try {
194
194
  mkdirSync(dest, { recursive: true });
@@ -209,8 +209,8 @@ function snapshotPlanArtifacts(dest) {
209
209
 
210
210
  function snapshotSubagentUsage(dest) {
211
211
  if (!dest) return;
212
- // pi-plans state root convention: <workdir>/.git/pi_plans/runs/<id>/subagents.jsonl
213
- const stateRoots = [path.join(process.cwd(), ".git", "pi_plans", "runs")];
212
+ // pi-plans state root convention: <workdir>/.git/pi-plans/runs/<id>/subagents.jsonl
213
+ const stateRoots = [path.join(process.cwd(), ".git", "pi-plans", "runs")];
214
214
  const totals = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, cost: 0, children: 0 };
215
215
  try {
216
216
  for (const root of stateRoots) {
@@ -1,7 +1,8 @@
1
1
  #!/usr/bin/env node
2
2
 
3
3
  import { spawnSync } from "node:child_process";
4
- import { existsSync, readdirSync } from "node:fs";
4
+ import { existsSync, mkdtempSync, readdirSync } from "node:fs";
5
+ import { tmpdir } from "node:os";
5
6
  import { join, resolve } from "node:path";
6
7
 
7
8
  const testDir = resolve("tests");
@@ -10,6 +11,15 @@ if (!existsSync(testDir)) {
10
11
  process.exit(1);
11
12
  }
12
13
 
14
+ // Safety net (F-002): unless the caller pinned one, point the pi-plans
15
+ // GLOBAL config at a throwaway directory so concurrent test files can never
16
+ // clobber the developer's real ~/.pi/pi-plans/config.json. Individual suites
17
+ // may still override per-test with their own PI_PLANS_GLOBAL_DIR.
18
+ const env = { ...process.env };
19
+ if (!env.PI_PLANS_GLOBAL_DIR) {
20
+ env.PI_PLANS_GLOBAL_DIR = mkdtempSync(join(tmpdir(), "pi-plans-global-"));
21
+ }
22
+
13
23
  const tests = readdirSync(testDir)
14
24
  .filter((entry) => entry.endsWith(".test.ts"))
15
25
  .sort()
@@ -22,6 +32,7 @@ if (tests.length === 0) {
22
32
 
23
33
  const result = spawnSync(process.execPath, ["--experimental-strip-types", "--test", ...tests], {
24
34
  stdio: "inherit",
35
+ env,
25
36
  });
26
37
 
27
38
  if (result.error) {
@@ -17,7 +17,7 @@ const EXPECTED_SKILLS = new Set([
17
17
  "debug-and-plan",
18
18
  ]);
19
19
  const REQUIRED_REFERENCES = ["pi-planning-workflow.md", "plan-artifact-template.md", "state-and-config.md"];
20
- const REQUIRED_AGENTS = ["reviewer.md", "criticizer.md"];
20
+ const REQUIRED_AGENTS = ["reviewer.md"];
21
21
  const REQUIRED_TOOL_FILES = [
22
22
  "tools/plans.ts",
23
23
  "tools/ask-choice.ts",
@@ -26,6 +26,7 @@ const REQUIRED_TOOL_FILES = [
26
26
  "tools/execute-plan.ts",
27
27
  "tools/code-graph.ts",
28
28
  "src/state.ts",
29
+ "src/global-state.ts",
29
30
  "src/guard.ts",
30
31
  "src/plan.ts",
31
32
  "src/subagent.ts",
@@ -69,7 +70,7 @@ function validateSkill(dir: string): void {
69
70
  if (!description || description.length > 1024) fail(`${file}: invalid description length`);
70
71
  if (!/Use|MUST USE/.test(description)) fail(`${file}: description should include routing language`);
71
72
 
72
- const requiredPhrases = ["Auto-complete", "ask_choice", "refine", "language", "reviewer", "criticizer", ".git/pi_plans", "drawback"];
73
+ const requiredPhrases = ["Auto-complete", "ask_choice", "refine", "language", "reviewer", ".git/pi-plans", "drawback"];
73
74
  for (const phrase of requiredPhrases) {
74
75
  if (!text.includes(phrase)) fail(`${file}: missing required phrase ${phrase!}`);
75
76
  }
@@ -77,14 +78,24 @@ function validateSkill(dir: string): void {
77
78
 
78
79
  function validateDefaultConfig(): void {
79
80
  const source = fs.readFileSync(path.join(ROOT, "src", "state.ts"), "utf8");
80
- if (!source.includes('"pi-plans-reviewer"') || !source.includes('"pi-plans-criticizer"')) {
81
- fail("src/state.ts: name_prefix defaults missing");
82
- }
83
- if (source.includes("effort")) fail("src/state.ts: per-role effort must not exist");
84
- if (!source.includes('"delegated-subagent"')) fail("src/state.ts: delegated-subagent default missing");
85
81
  if (!source.includes('artifact_root: DEFAULT_ARTIFACT_ROOT')) fail("src/state.ts: artifact_root default missing");
86
82
  if (!source.includes('artifact_root_source: "unset"')) fail("src/state.ts: artifact_root_source default missing");
87
83
  if (!source.includes('artifact_root_updated_at: null')) fail("src/state.ts: artifact_root_updated_at default missing");
84
+ // The reviewer role defaults live in the GLOBAL config module (v0.7.0):
85
+ // one reviewer, one file, shared across workspaces.
86
+ const globalSource = fs.readFileSync(path.join(ROOT, "src", "global-state.ts"), "utf8");
87
+ if (!globalSource.includes('"pi-plans-reviewer"')) {
88
+ fail("src/global-state.ts: reviewer name_prefix default missing");
89
+ }
90
+ if (!globalSource.includes('"delegated-subagent"')) {
91
+ fail("src/global-state.ts: delegated-subagent default missing");
92
+ }
93
+ if (!globalSource.includes('thinking_level')) {
94
+ fail("src/global-state.ts: reviewer thinking_level field missing");
95
+ }
96
+ if (!globalSource.includes("PI_PLANS_GLOBAL_DIR")) {
97
+ fail("src/global-state.ts: PI_PLANS_GLOBAL_DIR override missing");
98
+ }
88
99
  }
89
100
 
90
101
  interface PackageJson {
@@ -235,8 +246,8 @@ function main(): void {
235
246
  for (const ref of REQUIRED_REFERENCES) {
236
247
  const file = path.join(ROOT, "references", ref);
237
248
  if (!fs.existsSync(file)) fail(`missing reference ${ref}`);
238
- if (!fs.readFileSync(file, "utf8").includes(".git/pi_plans")) {
239
- fail(`${ref}: missing .git/pi_plans state location`);
249
+ if (!fs.readFileSync(file, "utf8").includes(".git/pi-plans")) {
250
+ fail(`${ref}: missing .git/pi-plans state location`);
240
251
  }
241
252
  }
242
253
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: debug-and-plan
3
- description: Diagnose failures before creating a Pi plan. MUST USE for bugs, CI failures, test failures, regressions, incidents, broken behavior, root cause, RCA, or debug-why requests before deciding whether to plan; preserve language, reviewer, and criticizer settings in `.git/pi_plans/config.json`; exclude ordinary feature planning, direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
3
+ description: Diagnose failures before creating a Pi plan. MUST USE for bugs, CI failures, test failures, regressions, incidents, broken behavior, root cause, RCA, or debug-why requests before deciding whether to plan; preserve the workspace language settings in `.git/pi-plans/config.json` and the reviewer role in the global config (`~/.pi/pi-plans/config.json`); exclude ordinary feature planning, direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
4
4
  ---
5
5
 
6
6
  # Debug And Plan
@@ -9,7 +9,7 @@ Use this skill for problem or failure inputs that need diagnosis before planning
9
9
 
10
10
  ## Pi Setup
11
11
 
12
- Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer/criticizer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi_plans/`. Ask every question with the `ask_choice` tool (batch related questions into one `questions: [...]` form call, 2-8 items; scope/handoff stays single-question with `autoComplete: false`); run refinement rounds with the `refine` tool.
12
+ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool (batch related questions into one `questions: [...]` form call, 2-8 items; scope/handoff stays single-question with `autoComplete: false`); run refinement rounds with the `refine` tool.
13
13
 
14
14
  ## Diagnostic Workflow
15
15
 
@@ -32,4 +32,4 @@ Do not ask the user to choose the level unless the evidence supports two materia
32
32
 
33
33
  ## PROBLEM_ANALYSIS.md
34
34
 
35
- After opt-in, create the selected planning run's `.git/pi_plans` state and public artifact directory (`plans` action `start-run`), then write `PROBLEM_ANALYSIS.md` before `PLAN_v1.md`. Include: original problem; symptoms and reproduction status; evidence inspected; RCA summary and 5 Whys (ending early with `unknown` when evidence stops); suspected root cause and confidence; planning skill selected and why; language, reviewer, and criticizer settings used; open diagnostic gaps the plan must address. Pass the original problem, RCA summary, evidence, and `PROBLEM_ANALYSIS.md` path into the selected planning skill.
35
+ After opt-in, create the selected planning run's `.git/pi-plans` state and public artifact directory (`plans` action `start-run`), then write `PROBLEM_ANALYSIS.md` before `PLAN_v1.md`. Include: original problem; symptoms and reproduction status; evidence inspected; RCA summary and 5 Whys (ending early with `unknown` when evidence stops); suspected root cause and confidence; planning skill selected and why; language, reviewer, and reviewer settings used; open diagnostic gaps the plan must address. Pass the original problem, RCA summary, evidence, and `PROBLEM_ANALYSIS.md` path into the selected planning skill.
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: plan-big
3
- description: Create a large Pi plan before implementation. Use for open-ended or high-risk repo efforts needing 10 or more planning questions, web research, concurrent reviewer or criticizer refinement, and refinement until convergence; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
3
+ description: Create a large Pi plan before implementation. Use for open-ended or high-risk repo efforts needing 10 or more planning questions, web research, concurrent reviewer refinement (findings plus questions), and refinement until convergence; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
4
4
  ---
5
5
 
6
6
  # Plan Big
@@ -9,7 +9,7 @@ Use this skill when the user wants a large, high-risk, or open-ended plan before
9
9
 
10
10
  ## Pi Setup
11
11
 
12
- Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer/criticizer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi_plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
12
+ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
13
13
 
14
14
  ## Depth Contract
15
15
 
@@ -18,7 +18,7 @@ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-
18
18
  - Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
19
19
  - Use web research during both brainstorming and refinement when outside facts, patterns, or ecosystem constraints matter, and cite sources in the plan. Optionally (never required) you may also search 1–2 named references — a paper (e.g. arXiv), an engineering blog post, or another repository — whose technique or measurements strengthen the plan's reasoning; cite their URLs in the plan's Evidence section. This optional reference search does not count against the planning-question limit and never requires downloading or analyzing material (that is plan-with-refs' job).
20
20
  - Ask the final scope confirmation, then write `PLAN_v1.md` per `../../references/plan-artifact-template.md`.
21
- - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one `Reviewer` round as three concurrent independent reviewers (`refine` with `reviewers: 3`, consolidated by the main agent per the shared workflow), then one `Criticizer` round; afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Beyond the default sequence, refine until convergence on high-priority findings, unresolved questions, or evidence gaps; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
21
+ - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one reviewer round as three concurrent independent reviewers (`refine` with `reviewers: 3`), each returning findings (`F-###`) and up to five questions (`Q-1..Q-5`); consolidate per the shared workflow, ask every question with `ask_choice`, record the answers, then revise. Afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Beyond the default sequence, refine until convergence on high-priority findings, unresolved questions, or evidence gaps; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
22
22
 
23
23
  ## Fit
24
24
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: plan-normal
3
- description: Create a researched Pi plan before implementation. Use for broad or risky repo changes needing 5 to 10 planning questions, web research, reviewer or criticizer refinement, and bounded refinement; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
3
+ description: Create a researched Pi plan before implementation. Use for broad or risky repo changes needing 5 to 10 planning questions, web research, reviewer refinement (findings plus questions), and bounded refinement; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
4
4
  ---
5
5
 
6
6
  # Plan Normal
@@ -9,7 +9,7 @@ Use this skill when the user wants a substantive plan before a repository change
9
9
 
10
10
  ## Pi Setup
11
11
 
12
- Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer/criticizer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi_plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
12
+ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
13
13
 
14
14
  ## Depth Contract
15
15
 
@@ -18,7 +18,7 @@ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-
18
18
  - Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
19
19
  - Use web research whenever outside library behavior, ecosystem precedent, UX convention, protocol semantics, or compatibility affects the recommendation (websearch skill when installed; otherwise `curl`/`gh` via bash), and cite sources in the plan. Optionally (never required) you may also search 1–2 named references — a paper (e.g. arXiv), an engineering blog post, or another repository — whose technique or measurements strengthen the plan's reasoning; cite their URLs in the plan's Evidence section. This optional reference search does not count against the planning-question limit and never requires downloading or analyzing material (that is plan-with-refs' job).
20
20
  - Ask the final scope confirmation, then write `PLAN_v1.md` per `../../references/plan-artifact-template.md`.
21
- - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one `Reviewer` round, then one `Criticizer` round; afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Up to five rounds total, continuing only for high-priority findings or unresolved criticizer questions; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
21
+ - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one reviewer round (findings `F-###` plus up to five questions `Q-1..Q-5` in the same output); ask every question with `ask_choice`, record the answers, then revise. Afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Up to five rounds total, continuing only for high-priority findings or unresolved reviewer questions; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
22
22
 
23
23
  ## Fit
24
24
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: plan-small
3
- description: Create a small Pi plan before implementation. Use for small scoped repo changes needing 1 to 3 planning questions and one criticizer refinement pass; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
3
+ description: Create a small Pi plan before implementation. Use for small scoped repo changes needing 1 to 3 planning questions and one reviewer refinement pass; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
4
4
  ---
5
5
 
6
6
  # Plan Small
@@ -9,15 +9,15 @@ Use this skill when the user wants a compact plan before a repository change.
9
9
 
10
10
  ## Pi Setup
11
11
 
12
- Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer/criticizer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi_plans/`. Ask every question with the `ask_choice` tool; run the refinement round with the `refine` tool.
12
+ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run the refinement round with the `refine` tool.
13
13
 
14
14
  ## Depth Contract
15
15
 
16
16
  - Inspect the target Git repo read-only before the first product question.
17
17
  - Ask 1 to 3 planning questions via `ask_choice` (recommended option first; the tool adds `Other` second-last and `Auto-complete` last). Every option you write carries a `description` of `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse — the user chooses by weighing what each option gains against what it costs. With 2-3 ready questions, batch them into ONE `questions: [...]` form call; a single question uses the classic `question` form.
18
18
  - Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
19
- - Ask the final scope confirmation, then write `PLAN_v1.md` under the artifact root (normally the configured workspace root, default `./docs/pi-plans/YYYY-MM-DD-topic/`) per `../../references/plan-artifact-template.md`.
20
- - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default: exactly one round, recommended mode `Criticizer`; afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round), then the `execute_plan` tool.
19
+ - Ask the final scope confirmation, then write `PLAN_v1.md` under the artifact root (normally the configured workspace root, default `./.git/pi-plans/plans/YYYY-MM-DD-topic/`) per `../../references/plan-artifact-template.md`.
20
+ - After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default: exactly one reviewer round (findings plus up to five questions); afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round), then the `execute_plan` tool.
21
21
 
22
22
  ## Fit
23
23