pi-plans 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +8 -15
- package/README.md +39 -37
- package/agents/execution-reviewer.md +40 -0
- package/agents/reviewer.md +12 -3
- package/index.ts +55 -58
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +50 -60
- package/references/plan-artifact-template.md +81 -60
- package/references/state-and-config.md +60 -44
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +39 -11
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +227 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +8 -3
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +303 -0
- package/src/exec.ts +1185 -924
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +45 -129
- package/src/resume.ts +5 -1
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/staleness.ts +53 -0
- package/src/state.ts +273 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +223 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +7 -54
- package/src/workflow-state.ts +76 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +210 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +402 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +771 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +41 -22
- package/tests/resume.test.ts +39 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +155 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +73 -90
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +55 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +41 -67
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/scripts/validate.ts
CHANGED
|
@@ -17,7 +17,12 @@ const EXPECTED_SKILLS = new Set([
|
|
|
17
17
|
"debug-and-plan",
|
|
18
18
|
]);
|
|
19
19
|
const REQUIRED_REFERENCES = ["pi-planning-workflow.md", "plan-artifact-template.md", "state-and-config.md"];
|
|
20
|
-
const REQUIRED_AGENTS = ["reviewer.md", "
|
|
20
|
+
const REQUIRED_AGENTS = ["reviewer.md", "execution-reviewer.md"];
|
|
21
|
+
// Root-level docs that must exist and must be entirely English (see AGENTS.md).
|
|
22
|
+
const REQUIRED_ROOT_DOCS = ["AGENTS.md", "CHANGELOG.md"];
|
|
23
|
+
// CJK ideographs plus full-width CJK punctuation. Ideographs alone are not
|
|
24
|
+
// enough: a file can read untranslated while carrying full-width punctuation.
|
|
25
|
+
const CJK_RE = /[ -〿一-鿿-]/;
|
|
21
26
|
const REQUIRED_TOOL_FILES = [
|
|
22
27
|
"tools/plans.ts",
|
|
23
28
|
"tools/ask-choice.ts",
|
|
@@ -26,6 +31,7 @@ const REQUIRED_TOOL_FILES = [
|
|
|
26
31
|
"tools/execute-plan.ts",
|
|
27
32
|
"tools/code-graph.ts",
|
|
28
33
|
"src/state.ts",
|
|
34
|
+
"src/global-state.ts",
|
|
29
35
|
"src/guard.ts",
|
|
30
36
|
"src/plan.ts",
|
|
31
37
|
"src/subagent.ts",
|
|
@@ -69,7 +75,7 @@ function validateSkill(dir: string): void {
|
|
|
69
75
|
if (!description || description.length > 1024) fail(`${file}: invalid description length`);
|
|
70
76
|
if (!/Use|MUST USE/.test(description)) fail(`${file}: description should include routing language`);
|
|
71
77
|
|
|
72
|
-
const requiredPhrases = ["Auto-complete", "ask_choice", "refine", "language", "reviewer", "
|
|
78
|
+
const requiredPhrases = ["Auto-complete", "ask_choice", "refine", "language", "reviewer", ".git/pi-plans", "drawback"];
|
|
73
79
|
for (const phrase of requiredPhrases) {
|
|
74
80
|
if (!text.includes(phrase)) fail(`${file}: missing required phrase ${phrase!}`);
|
|
75
81
|
}
|
|
@@ -77,14 +83,24 @@ function validateSkill(dir: string): void {
|
|
|
77
83
|
|
|
78
84
|
function validateDefaultConfig(): void {
|
|
79
85
|
const source = fs.readFileSync(path.join(ROOT, "src", "state.ts"), "utf8");
|
|
80
|
-
if (!source.includes('"pi-plans-reviewer"') || !source.includes('"pi-plans-criticizer"')) {
|
|
81
|
-
fail("src/state.ts: name_prefix defaults missing");
|
|
82
|
-
}
|
|
83
|
-
if (source.includes("effort")) fail("src/state.ts: per-role effort must not exist");
|
|
84
|
-
if (!source.includes('"delegated-subagent"')) fail("src/state.ts: delegated-subagent default missing");
|
|
85
86
|
if (!source.includes('artifact_root: DEFAULT_ARTIFACT_ROOT')) fail("src/state.ts: artifact_root default missing");
|
|
86
87
|
if (!source.includes('artifact_root_source: "unset"')) fail("src/state.ts: artifact_root_source default missing");
|
|
87
88
|
if (!source.includes('artifact_root_updated_at: null')) fail("src/state.ts: artifact_root_updated_at default missing");
|
|
89
|
+
// The reviewer role defaults live in the GLOBAL config module (v0.7.0):
|
|
90
|
+
// one reviewer, one file, shared across workspaces.
|
|
91
|
+
const globalSource = fs.readFileSync(path.join(ROOT, "src", "global-state.ts"), "utf8");
|
|
92
|
+
if (!globalSource.includes('"pi-plans-reviewer"')) {
|
|
93
|
+
fail("src/global-state.ts: reviewer name_prefix default missing");
|
|
94
|
+
}
|
|
95
|
+
if (!globalSource.includes('"delegated-subagent"')) {
|
|
96
|
+
fail("src/global-state.ts: delegated-subagent default missing");
|
|
97
|
+
}
|
|
98
|
+
if (!globalSource.includes('thinking_level')) {
|
|
99
|
+
fail("src/global-state.ts: reviewer thinking_level field missing");
|
|
100
|
+
}
|
|
101
|
+
if (!globalSource.includes("PI_PLANS_GLOBAL_DIR")) {
|
|
102
|
+
fail("src/global-state.ts: PI_PLANS_GLOBAL_DIR override missing");
|
|
103
|
+
}
|
|
88
104
|
}
|
|
89
105
|
|
|
90
106
|
interface PackageJson {
|
|
@@ -131,7 +147,7 @@ function validatePackageMetadata(): void {
|
|
|
131
147
|
if (!skills.has("./skills")) fail("package.json: pi.skills must include ./skills");
|
|
132
148
|
|
|
133
149
|
const files = new Set((pkg.files ?? []).map(normalizePackageEntry));
|
|
134
|
-
for (const required of ["README.md", "LICENSE", "CONTRIBUTING.md", "index.ts", "agents", "references", "scripts", "skills", "src", "tests", "tools"]) {
|
|
150
|
+
for (const required of ["README.md", "LICENSE", "CONTRIBUTING.md", "AGENTS.md", "index.ts", "agents", "references", "scripts", "skills", "src", "tests", "tools"]) {
|
|
135
151
|
if (!files.has(required)) fail(`package.json: files must include ${required}`);
|
|
136
152
|
}
|
|
137
153
|
for (const excluded of ["!scripts/bench/vendor", "!scripts/bench/results"]) {
|
|
@@ -154,6 +170,7 @@ const REQUIRED_PACK_ENTRIES = [
|
|
|
154
170
|
"README.md",
|
|
155
171
|
"LICENSE",
|
|
156
172
|
"CONTRIBUTING.md",
|
|
173
|
+
"AGENTS.md",
|
|
157
174
|
"index.ts",
|
|
158
175
|
"package.json",
|
|
159
176
|
"agents/reviewer.md",
|
|
@@ -235,8 +252,8 @@ function main(): void {
|
|
|
235
252
|
for (const ref of REQUIRED_REFERENCES) {
|
|
236
253
|
const file = path.join(ROOT, "references", ref);
|
|
237
254
|
if (!fs.existsSync(file)) fail(`missing reference ${ref}`);
|
|
238
|
-
if (!fs.readFileSync(file, "utf8").includes(".git/
|
|
239
|
-
fail(`${ref}: missing .git/
|
|
255
|
+
if (!fs.readFileSync(file, "utf8").includes(".git/pi-plans")) {
|
|
256
|
+
fail(`${ref}: missing .git/pi-plans state location`);
|
|
240
257
|
}
|
|
241
258
|
}
|
|
242
259
|
|
|
@@ -252,13 +269,24 @@ function main(): void {
|
|
|
252
269
|
if (!fs.existsSync(path.join(ROOT, tool))) fail(`missing ${tool}`);
|
|
253
270
|
}
|
|
254
271
|
|
|
272
|
+
for (const doc of REQUIRED_ROOT_DOCS) {
|
|
273
|
+
const file = path.join(ROOT, doc);
|
|
274
|
+
if (!fs.existsSync(file)) {
|
|
275
|
+
fail(`missing root doc ${doc}`);
|
|
276
|
+
continue;
|
|
277
|
+
}
|
|
278
|
+
if (CJK_RE.test(fs.readFileSync(file, "utf8"))) {
|
|
279
|
+
fail(`${doc}: must be entirely English (no CJK ideographs or full-width CJK punctuation)`);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
255
283
|
if (!fs.existsSync(path.join(ROOT, "index.ts"))) fail("missing index.ts");
|
|
256
284
|
|
|
257
285
|
validateDefaultConfig();
|
|
258
286
|
validatePlansTool();
|
|
259
287
|
validatePackageMetadata();
|
|
260
288
|
validatePackageArtifact();
|
|
261
|
-
console.log(`validated ${EXPECTED_SKILLS.size} skills, ${REQUIRED_REFERENCES.length} references, ${REQUIRED_AGENTS.length} agents, ${REQUIRED_TOOL_FILES.length} tools`);
|
|
289
|
+
console.log(`validated ${EXPECTED_SKILLS.size} skills, ${REQUIRED_REFERENCES.length} references, ${REQUIRED_AGENTS.length} agents, ${REQUIRED_TOOL_FILES.length} tools, ${REQUIRED_ROOT_DOCS.length} docs`);
|
|
262
290
|
}
|
|
263
291
|
|
|
264
292
|
main();
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: debug-and-plan
|
|
3
|
-
description: Diagnose failures before creating a Pi plan. MUST USE for bugs, CI failures, test failures, regressions, incidents, broken behavior, root cause, RCA, or debug-why requests before deciding whether to plan; preserve
|
|
3
|
+
description: Diagnose failures before creating a Pi plan. MUST USE for bugs, CI failures, test failures, regressions, incidents, broken behavior, root cause, RCA, or debug-why requests before deciding whether to plan; preserve the workspace language settings in `.git/pi-plans/config.json` and the reviewer role in the global config (`~/.pi/pi-plans/config.json`); exclude ordinary feature planning, direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Debug And Plan
|
|
@@ -9,7 +9,7 @@ Use this skill for problem or failure inputs that need diagnosis before planning
|
|
|
9
9
|
|
|
10
10
|
## Pi Setup
|
|
11
11
|
|
|
12
|
-
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer
|
|
12
|
+
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool (batch related questions into one `questions: [...]` form call, 2-8 items; scope/handoff stays single-question with `autoComplete: false`); run refinement rounds with the `refine` tool.
|
|
13
13
|
|
|
14
14
|
## Diagnostic Workflow
|
|
15
15
|
|
|
@@ -32,4 +32,4 @@ Do not ask the user to choose the level unless the evidence supports two materia
|
|
|
32
32
|
|
|
33
33
|
## PROBLEM_ANALYSIS.md
|
|
34
34
|
|
|
35
|
-
After opt-in, create the selected planning run's `.git/
|
|
35
|
+
After opt-in, create the selected planning run's `.git/pi-plans` state and public artifact directory (`plans` action `start-run`), then write `PROBLEM_ANALYSIS.md` before `PLAN_v1.md`. Include: original problem; symptoms and reproduction status; evidence inspected; RCA summary and 5 Whys (ending early with `unknown` when evidence stops); suspected root cause and confidence; planning skill selected and why; language, reviewer, and reviewer settings used; open diagnostic gaps the plan must address. Pass the original problem, RCA summary, evidence, and `PROBLEM_ANALYSIS.md` path into the selected planning skill.
|
package/skills/plan-big/SKILL.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: plan-big
|
|
3
|
-
description: Create a large Pi plan before implementation. Use for open-ended or high-risk repo efforts needing 10 or more planning questions, web research, concurrent reviewer
|
|
3
|
+
description: Create a large Pi plan before implementation. Use for open-ended or high-risk repo efforts needing 10 or more planning questions, web research, concurrent reviewer refinement (findings plus questions), and refinement until convergence; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Plan Big
|
|
@@ -9,7 +9,7 @@ Use this skill when the user wants a large, high-risk, or open-ended plan before
|
|
|
9
9
|
|
|
10
10
|
## Pi Setup
|
|
11
11
|
|
|
12
|
-
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer
|
|
12
|
+
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
|
|
13
13
|
|
|
14
14
|
## Depth Contract
|
|
15
15
|
|
|
@@ -18,7 +18,7 @@ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-
|
|
|
18
18
|
- Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
|
|
19
19
|
- Use web research during both brainstorming and refinement when outside facts, patterns, or ecosystem constraints matter, and cite sources in the plan. Optionally (never required) you may also search 1–2 named references — a paper (e.g. arXiv), an engineering blog post, or another repository — whose technique or measurements strengthen the plan's reasoning; cite their URLs in the plan's Evidence section. This optional reference search does not count against the planning-question limit and never requires downloading or analyzing material (that is plan-with-refs' job).
|
|
20
20
|
- Ask the final scope confirmation, then write `PLAN_v1.md` per `../../references/plan-artifact-template.md`.
|
|
21
|
-
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one
|
|
21
|
+
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one reviewer round as three concurrent independent reviewers (`refine` with `reviewers: 3`), each returning findings (`F-###`) and up to five questions (`Q-1..Q-5`); consolidate per the shared workflow, ask every question with `ask_choice`, record the answers, then revise. Afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Beyond the default sequence, refine until convergence on high-priority findings, unresolved questions, or evidence gaps; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
|
|
22
22
|
|
|
23
23
|
## Fit
|
|
24
24
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: plan-normal
|
|
3
|
-
description: Create a researched Pi plan before implementation. Use for broad or risky repo changes needing 5 to 10 planning questions, web research, reviewer
|
|
3
|
+
description: Create a researched Pi plan before implementation. Use for broad or risky repo changes needing 5 to 10 planning questions, web research, reviewer refinement (findings plus questions), and bounded refinement; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Plan Normal
|
|
@@ -9,7 +9,7 @@ Use this skill when the user wants a substantive plan before a repository change
|
|
|
9
9
|
|
|
10
10
|
## Pi Setup
|
|
11
11
|
|
|
12
|
-
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer
|
|
12
|
+
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run refinement rounds with the `refine` tool.
|
|
13
13
|
|
|
14
14
|
## Depth Contract
|
|
15
15
|
|
|
@@ -18,7 +18,7 @@ Read `../../references/pi-planning-workflow.md` and `../../references/state-and-
|
|
|
18
18
|
- Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
|
|
19
19
|
- Use web research whenever outside library behavior, ecosystem precedent, UX convention, protocol semantics, or compatibility affects the recommendation (websearch skill when installed; otherwise `curl`/`gh` via bash), and cite sources in the plan. Optionally (never required) you may also search 1–2 named references — a paper (e.g. arXiv), an engineering blog post, or another repository — whose technique or measurements strengthen the plan's reasoning; cite their URLs in the plan's Evidence section. This optional reference search does not count against the planning-question limit and never requires downloading or analyzing material (that is plan-with-refs' job).
|
|
20
20
|
- Ask the final scope confirmation, then write `PLAN_v1.md` per `../../references/plan-artifact-template.md`.
|
|
21
|
-
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one
|
|
21
|
+
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default sequence: one reviewer round (findings `F-###` plus up to five questions `Q-1..Q-5` in the same output); ask every question with `ask_choice`, record the answers, then revise. Afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question. Up to five rounds total, continuing only for high-priority findings or unresolved reviewer questions; surface at most five per round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
|
|
22
22
|
|
|
23
23
|
## Fit
|
|
24
24
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: plan-small
|
|
3
|
-
description: Create a small Pi plan before implementation. Use for small scoped repo changes needing 1 to 3 planning questions and one
|
|
3
|
+
description: Create a small Pi plan before implementation. Use for small scoped repo changes needing 1 to 3 planning questions and one reviewer refinement pass; exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Plan Small
|
|
@@ -9,15 +9,15 @@ Use this skill when the user wants a compact plan before a repository change.
|
|
|
9
9
|
|
|
10
10
|
## Pi Setup
|
|
11
11
|
|
|
12
|
-
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer
|
|
12
|
+
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool; run the refinement round with the `refine` tool.
|
|
13
13
|
|
|
14
14
|
## Depth Contract
|
|
15
15
|
|
|
16
16
|
- Inspect the target Git repo read-only before the first product question.
|
|
17
17
|
- Ask 1 to 3 planning questions via `ask_choice` (recommended option first; the tool adds `Other` second-last and `Auto-complete` last). Every option you write carries a `description` of `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse — the user chooses by weighing what each option gains against what it costs. With 2-3 ready questions, batch them into ONE `questions: [...]` form call; a single question uses the classic `question` form.
|
|
18
18
|
- Batch protocol (0.4.0): when a round has several questions, submit them together as one `ask_choice` call with `questions: [...]` (2-8 items, recommended option first per question) — the tool opens one tabbed multiple-choice form with a submit page instead of asking one at a time. After the batch returns, think about the answers, then follow up in later calls (batch again for 2+ related follow-ups; single `question` for one). `Esc` on the form returns the answered subset as partial answers — continue with what you got and re-ask only what matters. The final scope confirmation and the execution handoff are ALWAYS single-question calls with `autoComplete: false`; batches reject those questions.
|
|
19
|
-
- Ask the final scope confirmation, then write `PLAN_v1.md` under the artifact root (normally the configured workspace root, default
|
|
20
|
-
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default: exactly one round
|
|
19
|
+
- Ask the final scope confirmation, then write `PLAN_v1.md` under the artifact root (normally the configured workspace root, default `./.git/pi-plans/plans/YYYY-MM-DD-topic/`) per `../../references/plan-artifact-template.md`.
|
|
20
|
+
- After each plan version, ask the merged accept/execute question via `ask_choice` with `autoComplete: false` — never run `refine` unless the user picked another round at that question. Default: exactly one reviewer round (findings plus up to five questions); afterwards the recommended option is `✓ Accept & execute now` in the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round), then the `execute_plan` tool.
|
|
21
21
|
|
|
22
22
|
## Fit
|
|
23
23
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: plan-with-refs
|
|
3
|
-
description: Research references before creating a Pi plan. Use when repo-change planning needs downloaded projects, articles, papers, docs, per-reference analysis, adoption questions, language settings, and reviewer
|
|
3
|
+
description: Research references before creating a Pi plan. Use when repo-change planning needs downloaded projects, articles, papers, docs, per-reference analysis, adoption questions, language settings, and reviewer refinement (findings plus questions); exclude direct implementation-only, factual/explanation, trivial command-only, or explicit no-plan requests.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# Plan With Refs
|
|
@@ -9,22 +9,22 @@ Use this skill when external references must shape the plan before implementatio
|
|
|
9
9
|
|
|
10
10
|
## Pi Setup
|
|
11
11
|
|
|
12
|
-
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer
|
|
12
|
+
Read `../../references/pi-planning-workflow.md` and `../../references/state-and-config.md` — both normative — and follow their setup, state, `language`, and reviewer rules. Initialize workspace state with the `plans` tool (`action: "init"`); state lives in `.git/pi-plans/`. Ask every question with the `ask_choice` tool (batch related questions and adoption questions into one `questions: [...]` form call, 2-8 items; scope/handoff stays single-question with `autoComplete: false`); run refinement rounds with the `refine` tool.
|
|
13
13
|
|
|
14
14
|
## Required Reference Flow
|
|
15
15
|
|
|
16
16
|
1. Inspect the target Git repo read-only before external research so search terms match the actual codebase and constraints.
|
|
17
|
-
2. Create the `.git/
|
|
17
|
+
2. Create the `.git/pi-plans` run state and planning artifact directory once the topic is clear (`plans` action `start-run`).
|
|
18
18
|
3. Search proactively for related projects, articles, papers, docs, and prior art. Prefer a websearch skill when installed; otherwise use bash tools such as `curl` or `gh` when already available.
|
|
19
|
-
4. Before the first download, check `refs_root` in `.git/
|
|
19
|
+
4. Before the first download, check `refs_root` in `.git/pi-plans/config.json` (`plans` action `show`). If it is unset, ask exactly one `ask_choice` question — recommended `.git/pi-plans/refs/` (inside the git dir, never tracked), second `./refs/`, third `~/.cache/pi-plans/refs/` — with each option's `description` set to `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse, and persist with `plans` (`set-refs-root`); this question does not count against the planning-question limit. Then download at least 3 credible references across at least 2 distinct origins before writing `PLAN_v1.md`, under the configured refs root (one subdirectory per reference — `analyze_refs` requires directories). References are NOT limited to GitHub repositories: papers (e.g. arXiv), engineering blog posts, and documentation sites are first-class, and theoretical references count exactly as much as implementation references. A download is qualified per medium: repo = clone (full working tree); paper = the full text (arXiv HTML preferred, else the PDF with extracted text into `paper.txt`/`paper.md`; an abstract alone never qualifies); blog/docs site = the full-article readable markdown saved locally (single-page posts are fine — they ARE the source; only fragments or teasers fail). Landing pages, README-only snapshots, abstracts, package metadata, or curl-only fragments do not count when deeper source material is available.
|
|
20
20
|
5. For every reference, record source metadata and local path in `REF_ANALYSIS.md` and in the run's `refs.jsonl` (via `plans` action `record-ref`): title, URL, kind (`project` for repos, `paper` for papers, `article` for blog posts, `docs` for documentation sites), retrieval method, date accessed, local path, coverage, and evidence gaps.
|
|
21
21
|
6. For every reference, run the `analyze_refs` tool (required path — it replaces manual structured reads): one independent read-only subagent per reference deep-reads it and returns structured sections (Overview / Key Mechanisms And Design Tradeoffs / Adoptable Ideas For The Target Repo / Pitfalls And Anti-Patterns / Evidence Citations / Coverage / Evidence Gaps). Paste each analysis into `REF_ANALYSIS.md` and fill `coverage` and `gaps` in `refs.jsonl` via `plans` (`record-ref`) before asking adoption questions.
|
|
22
22
|
7. For every reference after analysis, ask at least 3 ref-specific adoption questions via `ask_choice` before using its ideas in `PLAN_v1.md`; each based on downloaded content, recommended option first, `Other` second-last, `Auto-complete` last (the tool appends both). Every option you write carries a `description` of `✓ <advantage> / ✗ <drawback>` in the configured language, kept terse — adoption choices trade a real gain against a real cost, so state both. Batch one reference's adoption questions into a single `questions: [...]` form call.
|
|
23
23
|
8. Block rather than pad if fewer than 3 credible references exist, unless the user explicitly narrows the topic or waives the minimum. `Auto-complete` cannot grant this waiver.
|
|
24
|
-
9. Continue with big-plan depth: at least 10 planning questions, required web research during brainstorming and refinement (`refine` `reviewers: 3` reviewer round
|
|
24
|
+
9. Continue with big-plan depth: at least 10 planning questions, required web research during brainstorming and refinement (`refine` `reviewers: 3` reviewer round carrying findings and questions), no refinement limit, at most five high-priority comments or questions per refinement round. Then the merged accept/execute question (ask_choice with `autoComplete: false`: ✓ Accept & execute now / Accept, don't execute yet / another round) and the `execute_plan` tool.
|
|
25
25
|
|
|
26
26
|
## REF_ANALYSIS.md
|
|
27
27
|
|
|
28
|
-
Include: original request and repo evidence that shaped the search; attempted queries and selection criteria; references selected and rejected; configured refs root and local download paths; the per-reference `analyze_refs` structured analyses (pasted verbatim, one section per reference); adoption questions and recorded answers; accepted ideas, rejected ideas, and reasons; evidence gaps and user-granted waivers; language, reviewer, and
|
|
28
|
+
Include: original request and repo evidence that shaped the search; attempted queries and selection criteria; references selected and rejected; configured refs root and local download paths; the per-reference `analyze_refs` structured analyses (pasted verbatim, one section per reference); adoption questions and recorded answers; accepted ideas, rejected ideas, and reasons; evidence gaps and user-granted waivers; language, reviewer, and reviewer settings used.
|
|
29
29
|
|
|
30
30
|
Reference ideas are not eligible for `PLAN_v1.md` until their adoption question answers are recorded.
|
package/skills/planning/SKILL.md
CHANGED
|
@@ -19,4 +19,4 @@ Use this skill when a task is planning-related but the right specialist is not o
|
|
|
19
19
|
|
|
20
20
|
## Pi Setup
|
|
21
21
|
|
|
22
|
-
Use the same language, `ask_choice`, `refine`, reviewer,
|
|
22
|
+
Use the same language, `ask_choice`, `refine`, reviewer, Auto-complete, and `.git/pi-plans` rules as the selected specialist skill and the shared workflow — including the per-option `✓ <advantage> / ✗ <drawback>` descriptions, which every option you write carries in the configured language, kept terse. Questions prefer the 0.4.0 batch form: one `ask_choice` call with `questions: [...]` (2-8) opens a tabbed multiple-choice form; scope confirmation and execution handoff stay single-question with `autoComplete: false`.
|
package/src/ask-form.ts
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
*
|
|
4
4
|
* The form is a tabbed dialog opened via ctx.ui.custom: one tab per question
|
|
5
5
|
* (options always visible in the first frame — options-first fit contract from
|
|
6
|
-
*
|
|
6
|
+
* a guided multi-question questionnaire) plus a final submit page listing every Q/A, and a
|
|
7
7
|
* custom-answer row per tab that switches into a Focusable single-line input
|
|
8
8
|
* (CURSOR_MARKER + hardware cursor so zh-Hans IME composition works).
|
|
9
9
|
*
|
|
@@ -69,7 +69,7 @@ export interface FormState {
|
|
|
69
69
|
custom: (string | null)[];
|
|
70
70
|
/** Whether the user explicitly confirmed an answer on tab i (Enter on an
|
|
71
71
|
* option row or a committed custom answer). The cursor `selection` alone
|
|
72
|
-
* never flips this —
|
|
72
|
+
* never flips this — the ■/□ chips read this field. */
|
|
73
73
|
confirmed: boolean[];
|
|
74
74
|
/** Current tab: 0..N-1 = question tabs, N = submit page. */
|
|
75
75
|
tab: number;
|
|
@@ -86,7 +86,7 @@ export function createFormState(questions: FormQuestion[], lang: UiLanguage = "e
|
|
|
86
86
|
// Cursor position per tab: pre-positioned on the recommended option
|
|
87
87
|
// where present. This is ONLY the highlight — an answer counts only
|
|
88
88
|
// after the user confirms it (Enter/custom submit), tracked in
|
|
89
|
-
// `confirmed`
|
|
89
|
+
// `confirmed` semantics: chips show □ until answered.
|
|
90
90
|
selection: questions.map((q) => q.options.findIndex((o) => o.recommended === true)),
|
|
91
91
|
confirmed: questions.map(() => false),
|
|
92
92
|
custom: questions.map(() => null),
|
|
@@ -286,7 +286,7 @@ export const FORM_ROWS_RESERVE = 4;
|
|
|
286
286
|
export const FORM_FALLBACK_ROWS = 30;
|
|
287
287
|
|
|
288
288
|
/**
|
|
289
|
-
*
|
|
289
|
+
* themed question-tab frame: accent borders, tab chips with a
|
|
290
290
|
* selectedBg background on the active chip (■ answered / □ pending), colored
|
|
291
291
|
* question line, dim key hints. Under a `rows` budget it degrades richest
|
|
292
292
|
* first — descriptions, then blank separators, then the borders (the chips
|
package/src/auditor.ts
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Execution reviewer (v0.8): when every task
|
|
3
|
+
* reaches a terminal state, an independent read-only subagent verifies the
|
|
4
|
+
* plan's verification checks against the worktree. Failed checks roll their
|
|
5
|
+
* covered tasks back to pending (exclusively inside the review flow). The
|
|
6
|
+
* loop is bounded at REVIEW_MAX_ROUNDS committed rounds; exhaustion pauses
|
|
7
|
+
* the run for the user in every mode (fail-closed, never a hang and never a
|
|
8
|
+
* silent stop) — only an explicit /plans-execute confirmation grants a fresh
|
|
9
|
+
* budget.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
13
|
+
import * as fs from "node:fs";
|
|
14
|
+
import * as path from "node:path";
|
|
15
|
+
import { auditableChecks, skippedPassCheckIds, type TaskView } from "./tasks.ts";
|
|
16
|
+
import type { CheckItem } from "./plan.ts";
|
|
17
|
+
import { messaging } from "./messaging.ts";
|
|
18
|
+
|
|
19
|
+
/** Budget cap: committed rounds per user-granted budget. Discarded (fingerprint-changed) attempts do not count. */
|
|
20
|
+
export const REVIEW_MAX_ROUNDS = 5;
|
|
21
|
+
|
|
22
|
+
/** Legacy alias for one release: checkpoints and old builds still know this name. */
|
|
23
|
+
export const AUDIT_MAX_ROUNDS = REVIEW_MAX_ROUNDS;
|
|
24
|
+
|
|
25
|
+
export interface AuditOutcome {
|
|
26
|
+
round: number;
|
|
27
|
+
passed: string[];
|
|
28
|
+
failed: string[];
|
|
29
|
+
/** Checks whose report carried no readable verdict, or whose evidence was
|
|
30
|
+
* inconclusive. Never treated as `failed`: see src/exec.ts. */
|
|
31
|
+
undeterminable: string[];
|
|
32
|
+
report: string;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Parse the verdict lines of an audit report against the checks that are
|
|
36
|
+
* still pending. A pending check with no readable verdict is
|
|
37
|
+
* `undeterminable`, never `failed` — the caller must be able to tell "the
|
|
38
|
+
* work is wrong" apart from "I could not read the answer". */
|
|
39
|
+
export interface ParsedAudit {
|
|
40
|
+
passed: string[];
|
|
41
|
+
failed: string[];
|
|
42
|
+
undeterminable: string[];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/** Checks this round must judge: auditable (covering at least one task) and
|
|
46
|
+
* not already done. The brief and the coverage self-check both use this, so a
|
|
47
|
+
* well-formed round-2 report that omits an already-done check is not mistaken
|
|
48
|
+
* for a contract violation. */
|
|
49
|
+
export function auditablePendingChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
|
|
50
|
+
return auditableChecks(checklist, tasks).filter((item) => !item.done);
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** Build the audit brief for the read-only subagent. Exported for tests. */
|
|
54
|
+
export function buildAuditTask(planPath: string, checklist: CheckItem[], tasks: TaskView[], round: number): string {
|
|
55
|
+
const checks = auditablePendingChecks(checklist, tasks)
|
|
56
|
+
.map((check) => `- \`${check.id}\`: ${check.text}`)
|
|
57
|
+
.join("\n");
|
|
58
|
+
return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks.
|
|
59
|
+
|
|
60
|
+
Target plan: ${planPath} (audit round ${round})
|
|
61
|
+
|
|
62
|
+
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
|
|
63
|
+
|
|
64
|
+
Evidence: inspect the repository with read, grep, find, ls, and targeted commands (bash is not granted — rely on the read tools) before judging each check. Tests may be referenced from their recorded evidence; do not re-run them.
|
|
65
|
+
|
|
66
|
+
Checks to verify (only these; checks covering no task and checks already satisfied in an earlier round are excluded):
|
|
67
|
+
|
|
68
|
+
${checks}
|
|
69
|
+
|
|
70
|
+
Output: Markdown with exactly one section per check, in the order above:
|
|
71
|
+
|
|
72
|
+
- \`VC-###\` — verdict: pass | fail | undeterminable; evidence: <repo path/command proving it>; note: <one line>.
|
|
73
|
+
|
|
74
|
+
Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence.`;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Parse the audit subagent's verdict lines. Exported for tests. */
|
|
78
|
+
export function parseAuditReport(report: string, pendingVcIds: string[]): ParsedAudit {
|
|
79
|
+
const passed: string[] = [];
|
|
80
|
+
const failed: string[] = [];
|
|
81
|
+
const known = new Set(pendingVcIds.map((id) => id.toUpperCase()));
|
|
82
|
+
// The auditor writes Markdown, so a verdict may carry emphasis
|
|
83
|
+
// (`**pass**`, `*pass*`, `_pass_`, `` `pass` ``). Requiring a bare token
|
|
84
|
+
// silently discarded such verdicts, and the caller's fail-closed rule then
|
|
85
|
+
// marked every check failed and rolled the whole run back -- reporting
|
|
86
|
+
// correct work as failure. Tolerate the markers; \b keeps `passed` and
|
|
87
|
+
// `passing` from matching.
|
|
88
|
+
for (const match of report.matchAll(/`?(VC-\d+)`?[^\n]*?verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/gi)) {
|
|
89
|
+
const id = match[1].toUpperCase();
|
|
90
|
+
if (!known.has(id)) continue;
|
|
91
|
+
const verdict = match[2].toLowerCase();
|
|
92
|
+
if (verdict === "pass") passed.push(id);
|
|
93
|
+
else if (verdict === "fail") failed.push(id);
|
|
94
|
+
}
|
|
95
|
+
// A check with conflicting verdicts resolves to fail: the reader saw both.
|
|
96
|
+
for (const id of [...new Set(passed)]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
|
|
97
|
+
const undecided = new Set(known);
|
|
98
|
+
for (const id of [...passed, ...failed]) undecided.delete(id);
|
|
99
|
+
return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided] };
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** Pure decision core: classify a parsed report into the audit outcome. It
|
|
103
|
+
* touches neither the checklist nor the task tree — the caller owns writing
|
|
104
|
+
* `done` and computing the rollback set, so there is exactly one authority for
|
|
105
|
+
* both. Exported for tests. */
|
|
106
|
+
export function applyAuditOutcome(round: number, parsed: ParsedAudit, report: string): AuditOutcome {
|
|
107
|
+
return {
|
|
108
|
+
round,
|
|
109
|
+
passed: parsed.passed,
|
|
110
|
+
failed: parsed.failed,
|
|
111
|
+
undeterminable: parsed.undeterminable,
|
|
112
|
+
report,
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** Skipped-pass checks (all covered tasks skipped) pass without audit. */
|
|
117
|
+
export function presolvedCheckIds(checklist: CheckItem[], tasks: TaskView[]): string[] {
|
|
118
|
+
return skippedPassCheckIds(checklist, tasks);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Spawn the audit subagent and classify its report. Returns `{ cancelled: true }`
|
|
122
|
+
* when the child was aborted (session shutdown, stop, restore, tree switch) —
|
|
123
|
+
* such a round burns no budget and sends no wake; `null` when the subagent
|
|
124
|
+
* itself failed to run (treated as an all-undeterminable committed round; the
|
|
125
|
+
* caller counts it against the round cap). */
|
|
126
|
+
export type AuditRoundResult = AuditOutcome | { cancelled: true } | null;
|
|
127
|
+
|
|
128
|
+
export async function runCompletionAudit(
|
|
129
|
+
ctx: ExtensionContext,
|
|
130
|
+
opts: {
|
|
131
|
+
planPath: string;
|
|
132
|
+
checklist: CheckItem[];
|
|
133
|
+
tasks: TaskView[];
|
|
134
|
+
round: number;
|
|
135
|
+
model?: string;
|
|
136
|
+
thinkingLevel?: string;
|
|
137
|
+
timeoutMs?: number;
|
|
138
|
+
signal?: AbortSignal;
|
|
139
|
+
onProgress?: (event: import("./subagent.ts").SubagentProgressEvent) => void;
|
|
140
|
+
},
|
|
141
|
+
): Promise<AuditRoundResult> {
|
|
142
|
+
const { runPiSubagent } = await import("./subagent.ts");
|
|
143
|
+
const pending = auditablePendingChecks(opts.checklist, opts.tasks);
|
|
144
|
+
const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round);
|
|
145
|
+
// agents/auditor.md, not agents/reviewer.md: the reviewer prompt mandates a
|
|
146
|
+
// plan-review shape (`## Findings` / `## Questions`, F-###) and never says
|
|
147
|
+
// "verdict", so auditing under it produced reports this parser could not
|
|
148
|
+
// read at all -- every check then fell through to fail-closed.
|
|
149
|
+
// v0.8: a missing agent definition is a hard error — the silent inline
|
|
150
|
+
// fallback once swapped in a minimal prompt that produced unparsable
|
|
151
|
+
// reports and burned whole audit budgets as undeterminable.
|
|
152
|
+
const agentPrompt = fs.readFileSync(new URL("../agents/execution-reviewer.md", import.meta.url), "utf8");
|
|
153
|
+
const result = await runPiSubagent({
|
|
154
|
+
systemPrompt: agentPrompt,
|
|
155
|
+
task,
|
|
156
|
+
cwd: ctx.cwd,
|
|
157
|
+
model: opts.model,
|
|
158
|
+
thinkingLevel: opts.thinkingLevel,
|
|
159
|
+
timeoutMs: opts.timeoutMs,
|
|
160
|
+
tools: ["read", "grep", "find", "ls"],
|
|
161
|
+
signal: opts.signal,
|
|
162
|
+
onProgress: opts.onProgress,
|
|
163
|
+
});
|
|
164
|
+
if (!result.ok) {
|
|
165
|
+
if (result.cancelled === true) return { cancelled: true };
|
|
166
|
+
return null;
|
|
167
|
+
}
|
|
168
|
+
const parsed = parseAuditReport(result.output, pending.map((item) => item.id));
|
|
169
|
+
messaging().appendEntry("pi-plans-audit", {
|
|
170
|
+
planPath: opts.planPath,
|
|
171
|
+
round: opts.round,
|
|
172
|
+
passed: parsed.passed,
|
|
173
|
+
failed: parsed.failed,
|
|
174
|
+
undeterminable: parsed.undeterminable,
|
|
175
|
+
});
|
|
176
|
+
return applyAuditOutcome(opts.round, parsed, result.output);
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/** Persist one review-round attempt under `<run-dir>/execution-review/`.
|
|
180
|
+
* The attempt index (not the budget round) names the file, so a discarded
|
|
181
|
+
* attempt's evidence survives its re-run: round-<budgetRound>-attempt-<k>.md.
|
|
182
|
+
* These files are the only cross-session record of what a round saw — the
|
|
183
|
+
* in-flight marker itself is memory-only. Best-effort: an unwritable run dir
|
|
184
|
+
* must never fail the loop itself. */
|
|
185
|
+
export function writeReviewRoundReport(
|
|
186
|
+
runDir: string,
|
|
187
|
+
entry: {
|
|
188
|
+
budgetRound: number;
|
|
189
|
+
attempt: number;
|
|
190
|
+
outcome: "passed" | "failed" | "undeterminable" | "discarded" | "spawn-failed" | "cancelled";
|
|
191
|
+
passed: string[];
|
|
192
|
+
failed: string[];
|
|
193
|
+
undeterminable: string[];
|
|
194
|
+
discardedReason?: string;
|
|
195
|
+
fingerprintCaptured?: string;
|
|
196
|
+
fingerprintFound?: string;
|
|
197
|
+
coveredTaskIds: string[];
|
|
198
|
+
report: string;
|
|
199
|
+
},
|
|
200
|
+
): string | null {
|
|
201
|
+
try {
|
|
202
|
+
const dir = path.join(runDir, "execution-review");
|
|
203
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
204
|
+
const file = path.join(dir, `round-${entry.budgetRound}-attempt-${entry.attempt}.md`);
|
|
205
|
+
const lines = [
|
|
206
|
+
`# Execution review round ${entry.budgetRound} (attempt ${entry.attempt})`,
|
|
207
|
+
"",
|
|
208
|
+
`- outcome: ${entry.outcome}`,
|
|
209
|
+
`- passed: ${entry.passed.join(", ") || "(none)"}`,
|
|
210
|
+
`- failed: ${entry.failed.join(", ") || "(none)"}`,
|
|
211
|
+
`- undeterminable: ${entry.undeterminable.join(", ") || "(none)"}`,
|
|
212
|
+
...(entry.discardedReason ? [`- discarded: ${entry.discardedReason}`] : []),
|
|
213
|
+
`- fingerprint (captured): ${entry.fingerprintCaptured ?? "(n/a)"}`,
|
|
214
|
+
`- fingerprint (at resolve): ${entry.fingerprintFound ?? "(n/a)"}`,
|
|
215
|
+
`- covered tasks: ${entry.coveredTaskIds.join(", ") || "(none)"}`,
|
|
216
|
+
"",
|
|
217
|
+
"## Report",
|
|
218
|
+
"",
|
|
219
|
+
entry.report,
|
|
220
|
+
"",
|
|
221
|
+
];
|
|
222
|
+
fs.writeFileSync(file, lines.join("\n"), "utf8");
|
|
223
|
+
return file;
|
|
224
|
+
} catch {
|
|
225
|
+
return null;
|
|
226
|
+
}
|
|
227
|
+
}
|
package/src/auto-approve.ts
CHANGED
|
@@ -28,7 +28,7 @@ export function isAutoApproveEnabled(): boolean {
|
|
|
28
28
|
* must answer interactively) while a false negative would silently bypass a
|
|
29
29
|
* safety gate. Word-boundary anchored where short words could over-match.
|
|
30
30
|
* NOTE: package *installation inside a disposable benchmark container* is
|
|
31
|
-
* plan-lifecycle (the
|
|
31
|
+
* plan-lifecycle (the reviewer legitimately asks about installing deps);
|
|
32
32
|
* only externally-visible state (publish/deploy/merge/push/credentials/...)
|
|
33
33
|
* is hard-rejected. Install-permission waivers are still caught via
|
|
34
34
|
* "waiver". */
|