pi-plans 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +3 -3
- package/README.md +39 -37
- package/agents/reviewer.md +12 -3
- package/index.ts +42 -35
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +44 -60
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +59 -43
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +20 -9
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +692 -919
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +34 -128
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/state.ts +272 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +63 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +33 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +48 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/tools/refine.ts
CHANGED
|
@@ -1,20 +1,35 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `refine` tool — reviewer
|
|
3
|
-
*
|
|
2
|
+
* `refine` tool — reviewer refinement rounds via read-only Pi subagents
|
|
3
|
+
* with isolated context.
|
|
4
4
|
*
|
|
5
|
-
* Enforces the role
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* Enforces the reviewer role gates: the mode question stays agent-mediated
|
|
6
|
+
* (ask_choice), while first-use model confirmation pops native panels in
|
|
7
|
+
* TUI (v0.7.0): a /model-style searchable panel, then a /thinking-style
|
|
8
|
+
* effort panel, persisted to the GLOBAL reviewer config. Esc cancels the
|
|
9
|
+
* whole gate with a dedicated error (details.cancelled) — do not re-ask.
|
|
10
|
+
* The reviewer output carries findings AND questions (Q-###); the caller
|
|
11
|
+
* must ask every question with ask_choice before revising the plan.
|
|
8
12
|
*/
|
|
9
13
|
|
|
10
|
-
import {
|
|
14
|
+
import { Type } from "typebox";
|
|
11
15
|
import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
12
16
|
import { truncateHead } from "@earendil-works/pi-coding-agent";
|
|
13
17
|
import { Text } from "@earendil-works/pi-tui";
|
|
14
|
-
import { Type } from "typebox";
|
|
15
18
|
import * as fs from "node:fs";
|
|
16
19
|
import * as path from "node:path";
|
|
17
|
-
import {
|
|
20
|
+
import {
|
|
21
|
+
loadConfig,
|
|
22
|
+
normalizeWorkdir,
|
|
23
|
+
readActive,
|
|
24
|
+
recordSubagent,
|
|
25
|
+
resolveEffectiveReviewer,
|
|
26
|
+
resolveGlobalConfigPath,
|
|
27
|
+
resolveStateRootOrNull,
|
|
28
|
+
reviewerReady,
|
|
29
|
+
StateError,
|
|
30
|
+
} from "../src/state.ts";
|
|
31
|
+
import { runFirstUseFlow, firstUseCancelledError, firstUseTextGuidance, availableModels, findModel, type FirstUseOutcome, type RolePanelHost } from "../src/role-panels.ts";
|
|
32
|
+
import { roleModelLabel } from "../src/thinking-levels.ts";
|
|
18
33
|
import { uiLanguageFromTag, type UiLanguage } from "../src/ui-language.ts";
|
|
19
34
|
import type { SubagentUsage } from "../src/subagent.ts";
|
|
20
35
|
import { resolveActiveRun } from "../src/run-context.ts";
|
|
@@ -25,27 +40,20 @@ import {
|
|
|
25
40
|
reusableLaneOutputs,
|
|
26
41
|
startReviewRound,
|
|
27
42
|
} from "../src/workflow-state.ts";
|
|
28
|
-
import {
|
|
43
|
+
import { buildReviewerTask, reviewerLanes } from "../src/refine-prompts.ts";
|
|
29
44
|
import { graphBlockForRefiner } from "../src/code-graph/prompts.ts";
|
|
30
45
|
import { runPiSubagent, stripFrontmatter } from "../src/subagent.ts";
|
|
31
46
|
import { RefineOverlayController, refineOverlayContext } from "../src/refine-ui.ts";
|
|
32
47
|
|
|
33
48
|
|
|
34
49
|
const RefineParams = Type.Object({
|
|
35
|
-
role: StringEnum(["reviewer", "criticizer"] as const, { description: "Refinement role to run" }),
|
|
36
50
|
planPath: Type.String({ description: "Path to the PLAN_vN.md to review (absolute or relative to workdir)" }),
|
|
37
|
-
target: Type.Optional(
|
|
38
|
-
StringEnum(["plan", "implementation"] as const, {
|
|
39
|
-
description:
|
|
40
|
-
'Review target: "plan" (default) reviews the plan text; "implementation" reviews the implemented worktree against the plan\'s goals and acceptance criteria (post-execution amelioration).',
|
|
41
|
-
}),
|
|
42
|
-
),
|
|
43
51
|
focus: Type.Optional(Type.String({ description: "Specific concerns to direct the pass at" })),
|
|
44
52
|
reviewers: Type.Optional(
|
|
45
53
|
Type.Integer({
|
|
46
54
|
minimum: 1,
|
|
47
55
|
maximum: 3,
|
|
48
|
-
description: "Number of independent reviewer subagents (big plans: 3 for the concurrent round).
|
|
56
|
+
description: "Number of independent reviewer subagents (big plans: 3 for the concurrent round).",
|
|
49
57
|
}),
|
|
50
58
|
),
|
|
51
59
|
context: Type.Optional(
|
|
@@ -54,27 +62,52 @@ const RefineParams = Type.Object({
|
|
|
54
62
|
resumeRoundId: Type.Optional(
|
|
55
63
|
Type.String({
|
|
56
64
|
description:
|
|
57
|
-
'Round id to resume
|
|
65
|
+
'Round id to resume. Lanes already complete for this round in the run checkpoint are reused from their persisted outputs; only pending/failed/missing lanes run. Never reuse a round id across plan versions.',
|
|
58
66
|
}),
|
|
59
67
|
),
|
|
60
68
|
workdir: Type.Optional(Type.String({ description: "Target workspace; default current working directory" })),
|
|
61
69
|
});
|
|
62
70
|
|
|
63
|
-
function roleGateError(
|
|
71
|
+
function roleGateError(problem: "mode" | "confirm", guidance?: string): StateError {
|
|
64
72
|
if (problem === "mode") {
|
|
65
73
|
return new StateError(
|
|
66
|
-
`The
|
|
74
|
+
`The reviewer role mode is missing or invalid. Ask the role-setting question with ask_choice first: 1. Delegated subagent (recommended; read-only pi subprocess with isolated context) 2. Current session (run the pass yourself in this session) 3. Other 4. Auto-complete — then persist with the plans tool (set-role). The reviewer role lives in the global config (${resolveGlobalConfigPath()}).`,
|
|
67
75
|
);
|
|
68
76
|
}
|
|
69
77
|
return new StateError(
|
|
70
|
-
|
|
78
|
+
guidance ??
|
|
79
|
+
firstUseTextGuidance([], resolveGlobalConfigPath()),
|
|
71
80
|
);
|
|
72
81
|
}
|
|
73
82
|
|
|
83
|
+
/** First-use gate shared by the delegated spawn path: pop panels (TUI) or
|
|
84
|
+
* menus (hasUI non-TUI), persist on completion, cancel cleanly on Esc.
|
|
85
|
+
* Returns the role to use for THIS invocation, or throws. */
|
|
86
|
+
async function ensureReviewerReady(
|
|
87
|
+
toolName: string,
|
|
88
|
+
host: RolePanelHost,
|
|
89
|
+
role: { mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null },
|
|
90
|
+
): Promise<{ mode: string; model_selector: string | null; thinking_level: string | null; confirmed_at: string | null; name_prefix: string }> {
|
|
91
|
+
if (role.mode === "current-session" || reviewerReady(role as never)) return role as never;
|
|
92
|
+
let outcome: FirstUseOutcome = await runFirstUseFlow(host, role.thinking_level);
|
|
93
|
+
if (outcome.status === "confirmed" && outcome.model_selector !== null) {
|
|
94
|
+
// F-008: validate the freshly chosen selector against the registry when
|
|
95
|
+
// one is present, so a typo'd manual entry fails here, not at spawn.
|
|
96
|
+
if (availableModels(host).length > 0 && findModel(host, outcome.model_selector) === null) {
|
|
97
|
+
outcome = await runFirstUseFlow(host, outcome.role.thinking_level);
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
if (outcome.status === "cancelled") throw firstUseCancelledError(toolName);
|
|
101
|
+
if (outcome.status === "unavailable") {
|
|
102
|
+
throw roleGateError("confirm", firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath()));
|
|
103
|
+
}
|
|
104
|
+
if (outcome.role.model_selector === null) throw roleGateError("confirm", firstUseTextGuidance(availableModels(host), resolveGlobalConfigPath()));
|
|
105
|
+
return outcome.role;
|
|
106
|
+
}
|
|
107
|
+
|
|
74
108
|
function setupRefinementExecution(
|
|
75
109
|
ctx: ExtensionContext,
|
|
76
110
|
parentSignal: AbortSignal | undefined,
|
|
77
|
-
role: "reviewer" | "criticizer",
|
|
78
111
|
lanes: Array<{ id: string; label?: string }>,
|
|
79
112
|
modelLabel?: string,
|
|
80
113
|
lang: UiLanguage = "en",
|
|
@@ -84,7 +117,7 @@ function setupRefinementExecution(
|
|
|
84
117
|
if (parentSignal?.aborted) controller.abort();
|
|
85
118
|
else parentSignal?.addEventListener("abort", relayAbort, { once: true });
|
|
86
119
|
|
|
87
|
-
const overlay = ctx.mode === "tui" ? new RefineOverlayController(
|
|
120
|
+
const overlay = ctx.mode === "tui" ? new RefineOverlayController("reviewer", lanes, relayAbort, lang) : undefined;
|
|
88
121
|
overlay?.open(refineOverlayContext(ctx), modelLabel);
|
|
89
122
|
|
|
90
123
|
return {
|
|
@@ -97,20 +130,20 @@ function setupRefinementExecution(
|
|
|
97
130
|
};
|
|
98
131
|
}
|
|
99
132
|
|
|
100
|
-
export function registerRefineTool(
|
|
133
|
+
export function registerRefineTool(ext: ExtensionAPI, baseDir: string): void {
|
|
101
134
|
const agentsDir = path.join(baseDir, "agents");
|
|
102
135
|
|
|
103
|
-
const loadAgentPrompt = (
|
|
104
|
-
const file = path.join(agentsDir,
|
|
136
|
+
const loadAgentPrompt = (): string => {
|
|
137
|
+
const file = path.join(agentsDir, "reviewer.md");
|
|
105
138
|
return stripFrontmatter(fs.readFileSync(file, "utf8"));
|
|
106
139
|
};
|
|
107
140
|
|
|
108
|
-
|
|
141
|
+
ext.registerTool({
|
|
109
142
|
name: "refine",
|
|
110
143
|
label: "Refine",
|
|
111
144
|
description:
|
|
112
|
-
"Run a reviewer
|
|
113
|
-
promptSnippet: "Run reviewer
|
|
145
|
+
"Run a reviewer refinement round on a PLAN_vN.md via read-only Pi subagents. Each reviewer returns findings (F-###, severity, evidence, impact, fix, disposition) AND questions (Q-1..Q-5) that only the user can settle — after the round you MUST ask every question with ask_choice (one call per question or a batched form, in the configured language, stable questionIds) and record the answers before revising the plan. Use reviewers: 3 for concurrent reviewer rounds (big-plan review). Refuses to spawn until the reviewer mode is set and, for delegated-subagent, a concrete model is confirmed: first use pops native model/effort panels in TUI (persisted to the global reviewer config) instead of an ask_choice question.",
|
|
146
|
+
promptSnippet: "Run reviewer plan-refinement rounds",
|
|
114
147
|
parameters: RefineParams,
|
|
115
148
|
|
|
116
149
|
async execute(_toolCallId, params, signal, _onUpdate, ctx) {
|
|
@@ -122,19 +155,23 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
122
155
|
throw new StateError("no pi-plans state found; run the plans tool (action: init) first");
|
|
123
156
|
}
|
|
124
157
|
const config = loadConfig(root);
|
|
125
|
-
const roleConfig = config[params.role] as RoleConfig | undefined;
|
|
126
|
-
if (!roleConfig || (roleConfig.mode !== "delegated-subagent" && roleConfig.mode !== "current-session")) {
|
|
127
|
-
throw roleGateError(params.role, roleConfig, "mode");
|
|
128
|
-
}
|
|
129
|
-
if (roleConfig.confirmed_at === null) {
|
|
130
|
-
throw roleGateError(params.role, roleConfig, "confirm");
|
|
131
|
-
}
|
|
132
158
|
|
|
133
|
-
//
|
|
159
|
+
// F-005: cheap validations BEFORE any first-use panel, so a bad planPath
|
|
160
|
+
// never walks the user through two panels that would then be discarded.
|
|
134
161
|
const planPath = path.resolve(workdir, params.planPath.replace(/^@/, ""));
|
|
135
162
|
if (!fs.existsSync(planPath)) throw new StateError(`plan file not found: ${planPath}`);
|
|
136
163
|
const planText = fs.readFileSync(planPath, "utf8");
|
|
137
164
|
|
|
165
|
+
// Effective reviewer: global config first, legacy workspace block
|
|
166
|
+
// second — resolved read-only, never written here (F-001).
|
|
167
|
+
const { reviewer: initialRole } = resolveEffectiveReviewer(root);
|
|
168
|
+
if (initialRole.mode !== "delegated-subagent" && initialRole.mode !== "current-session") {
|
|
169
|
+
throw roleGateError("mode");
|
|
170
|
+
}
|
|
171
|
+
// Model/effort confirmation applies only to delegated-subagent
|
|
172
|
+
// (decision 10); current-session runs in this session with its model.
|
|
173
|
+
const roleConfig = await ensureReviewerReady("refine", ctx as unknown as RolePanelHost, initialRole);
|
|
174
|
+
|
|
138
175
|
const overlayLang = uiLanguageFromTag(config.language.tag);
|
|
139
176
|
// Record spawns against the active run when one exists.
|
|
140
177
|
const active = resolveActiveRun(ctx.sessionManager, workdir);
|
|
@@ -142,10 +179,10 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
142
179
|
if (!active) return;
|
|
143
180
|
try {
|
|
144
181
|
recordSubagent(workdir, active.run_id, {
|
|
145
|
-
role:
|
|
182
|
+
role: "reviewer",
|
|
146
183
|
name,
|
|
147
184
|
model: model ?? null,
|
|
148
|
-
|
|
185
|
+
thinking_level: roleConfig.mode === "current-session" ? null : roleConfig.thinking_level,
|
|
149
186
|
usage: usage
|
|
150
187
|
? { input: usage.input, output: usage.output, cache_read: usage.cacheRead, cache_write: usage.cacheWrite, cost: usage.cost }
|
|
151
188
|
: null,
|
|
@@ -155,36 +192,18 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
155
192
|
}
|
|
156
193
|
};
|
|
157
194
|
|
|
158
|
-
|
|
159
|
-
// I-004: durable round bookkeeping. Rounds start (or resume) in the
|
|
195
|
+
// Durable round bookkeeping. Rounds start (or resume) in the
|
|
160
196
|
// checkpoint BEFORE any lane spawns; successful outputs are persisted
|
|
161
|
-
// BEFORE the tool result returns
|
|
197
|
+
// BEFORE the tool result returns.
|
|
162
198
|
const checkpointLoad = active ? loadCheckpoint(workdir, active.run_id) : null;
|
|
163
199
|
const useCheckpoint = checkpointLoad?.status === "ok" ? checkpointLoad.checkpoint : null;
|
|
164
|
-
const roundId =
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
// round with an omitted `reviewers` reads the run's configured value
|
|
168
|
-
// from the checkpoint, so restarts/migrations never silently revert 2/3
|
|
169
|
-
// to 1. Explicit params always win; plan-review rounds are unchanged.
|
|
170
|
-
const configuredImplReviewers =
|
|
171
|
-
target === "implementation" && params.role === "reviewer"
|
|
172
|
-
? useCheckpoint?.implementationReview?.reviewerCount
|
|
173
|
-
: undefined;
|
|
174
|
-
const roundReviewerCount =
|
|
175
|
-
params.role === "reviewer"
|
|
176
|
-
? Math.min(3, Math.max(1, params.reviewers ?? configuredImplReviewers ?? 1))
|
|
177
|
-
: 1;
|
|
178
|
-
// F-001 (implementation review): the spec MUST carry lanes —
|
|
179
|
-
// reviewerLanes(count) for reviewer rounds, one lane for criticizer.
|
|
180
|
-
const roundLanes =
|
|
181
|
-
params.role === "reviewer"
|
|
182
|
-
? reviewerLanes(roundReviewerCount).map((lane) => ({ laneId: lane.id, lens: lane.lens ?? undefined }))
|
|
183
|
-
: [{ laneId: "criticizer" }];
|
|
200
|
+
const roundId = params.resumeRoundId ?? `plan-reviewer-r${Date.now().toString(36)}`;
|
|
201
|
+
const roundReviewerCount = Math.min(3, Math.max(1, params.reviewers ?? 1));
|
|
202
|
+
const roundLanes = reviewerLanes(roundReviewerCount).map((lane) => ({ laneId: lane.id, lens: lane.lens ?? undefined }));
|
|
184
203
|
const roundSpec = {
|
|
185
204
|
roundId,
|
|
186
|
-
role:
|
|
187
|
-
target,
|
|
205
|
+
role: "reviewer" as const,
|
|
206
|
+
target: "plan" as const,
|
|
188
207
|
reviewers: roundReviewerCount,
|
|
189
208
|
planPath,
|
|
190
209
|
focus: params.focus,
|
|
@@ -205,110 +224,43 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
205
224
|
/* the subagents ledger still records the spawn; resume treats the lane as unfinished */
|
|
206
225
|
}
|
|
207
226
|
};
|
|
208
|
-
const pickTask = (
|
|
209
|
-
|
|
210
|
-
return target === "implementation"
|
|
211
|
-
? buildImplementationReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context })
|
|
212
|
-
: buildReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context });
|
|
213
|
-
}
|
|
214
|
-
return target === "implementation"
|
|
215
|
-
? buildImplementationCriticizerTask({ planText, planPath, focus: params.focus, context: params.context })
|
|
216
|
-
: buildCriticizerTask({ planText, planPath, focus: params.focus, context: params.context });
|
|
217
|
-
};
|
|
227
|
+
const pickTask = (lens: string | null): string =>
|
|
228
|
+
buildReviewerTask({ planText, planPath, lens, focus: params.focus, context: params.context });
|
|
218
229
|
|
|
219
|
-
const systemPrompt = loadAgentPrompt(
|
|
230
|
+
const systemPrompt = loadAgentPrompt();
|
|
220
231
|
const graphEnabled = config.graph_enabled === true;
|
|
221
232
|
const subagentTools = graphEnabled ? ["read", "grep", "find", "ls", "code_graph"] : undefined;
|
|
222
233
|
const graphPrompt = graphBlockForRefiner(graphEnabled);
|
|
223
|
-
const
|
|
224
|
-
const
|
|
225
|
-
const modelLabel = model ?? "inherit";
|
|
234
|
+
const model = roleConfig.mode === "current-session" ? (ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : undefined) : roleConfig.model_selector ?? undefined;
|
|
235
|
+
const modelLabel = roleModelLabel(model ?? "inherit", roleConfig.mode === "current-session" ? null : roleConfig.thinking_level);
|
|
226
236
|
|
|
227
237
|
if (roleConfig.mode === "current-session") {
|
|
228
|
-
const task = pickTask(
|
|
238
|
+
const task = pickTask(null);
|
|
229
239
|
return {
|
|
230
240
|
content: [
|
|
231
241
|
{
|
|
232
242
|
type: "text",
|
|
233
|
-
text: `Role mode is current-session: perform the read-only
|
|
243
|
+
text: `Role mode is current-session: perform the read-only reviewer pass yourself, in this session, following this brief. Do not spawn anything. Then surface the findings and ask the Questions section with ask_choice before revising.\n\n${task}`,
|
|
234
244
|
},
|
|
235
245
|
],
|
|
236
|
-
details: { mode: "current-session",
|
|
246
|
+
details: { mode: "current-session", planPath },
|
|
237
247
|
};
|
|
238
248
|
}
|
|
239
249
|
|
|
240
|
-
|
|
241
|
-
const laneId = "criticizer";
|
|
242
|
-
const persisted = reusable[laneId];
|
|
243
|
-
if (persisted) {
|
|
244
|
-
return {
|
|
245
|
-
content: [
|
|
246
|
-
{
|
|
247
|
-
type: "text",
|
|
248
|
-
text: `${readReviewOutput(workdir, active!.run_id, persisted)}\n\n---\nReused the persisted criticizer result for round ${roundId} (no re-run). Ask each criticizer question with ask_choice (one call per question, in the configured language), record every answer, then revise the plan only after every question has an answer.`,
|
|
249
|
-
},
|
|
250
|
-
],
|
|
251
|
-
details: { mode: "delegated-subagent", role: params.role, planPath, target, roundId, reused: true },
|
|
252
|
-
};
|
|
253
|
-
}
|
|
254
|
-
const name = `${roleConfig.name_prefix}-criticizer-${Date.now().toString(36)}`;
|
|
255
|
-
const execution = setupRefinementExecution(ctx, signal, "criticizer", [{ id: name, label: "criticizer" }], modelLabel, overlayLang);
|
|
256
|
-
try {
|
|
257
|
-
const result = await runPiSubagent({
|
|
258
|
-
systemPrompt: `${systemPrompt}\n\n${graphPrompt}`,
|
|
259
|
-
task: pickTask("criticizer", null),
|
|
260
|
-
cwd: workdir,
|
|
261
|
-
model,
|
|
262
|
-
tools: subagentTools,
|
|
263
|
-
signal: execution.signal,
|
|
264
|
-
onProgress: (event) => execution.overlay?.update(name, event),
|
|
265
|
-
});
|
|
266
|
-
execution.overlay?.complete(name, result);
|
|
267
|
-
record(name, result.ok ? result.model ?? model : null, result.usage);
|
|
268
|
-
persistOutcome(laneId, result.ok ? { ok: true, output: result.output } : { ok: false, error: result.errorMessage });
|
|
269
|
-
if (!result.ok) {
|
|
270
|
-
throw new Error(
|
|
271
|
-
`criticizer subagent failed: ${result.errorMessage ?? "unknown error"}${result.stderr ? `\nstderr: ${result.stderr.slice(0, 2000)}` : ""}`,
|
|
272
|
-
);
|
|
273
|
-
}
|
|
274
|
-
return {
|
|
275
|
-
content: [
|
|
276
|
-
{
|
|
277
|
-
type: "text",
|
|
278
|
-
text: `${result.output}\n\n---\nAsk each criticizer question with ask_choice (one call per question, in the configured language, with a stable questionId per question), record every answer, then revise the plan only after every question has an answer. After the revision, record the boundary: plans record-checkpoint (checkpoint: { transition: "review-consolidated", roundId: "${roundId}" }).`,
|
|
279
|
-
},
|
|
280
|
-
],
|
|
281
|
-
details: { mode: "delegated-subagent", role: params.role, planPath, target, roundId, model: result.model ?? model },
|
|
282
|
-
};
|
|
283
|
-
} finally {
|
|
284
|
-
await execution.close();
|
|
285
|
-
}
|
|
286
|
-
}
|
|
287
|
-
|
|
288
|
-
const count = Math.min(3, Math.max(1, params.reviewers ?? configuredImplReviewers ?? 1));
|
|
250
|
+
const count = roundReviewerCount;
|
|
289
251
|
const lanes = reviewerLanes(count);
|
|
290
252
|
const jobs = lanes.map((lane) => {
|
|
291
253
|
const name = `${roleConfig.name_prefix}-${active?.run_id ?? "adhoc"}-${lane.id}`;
|
|
292
|
-
const task = pickTask(
|
|
254
|
+
const task = pickTask(lane.lens);
|
|
293
255
|
return { lane, name, task };
|
|
294
256
|
});
|
|
295
257
|
|
|
296
|
-
//
|
|
297
|
-
const roundsSlot = ctx.sessionManager as unknown as { __ameliorateRounds?: Map<string, number> };
|
|
298
|
-
const nextRound = (planPath: string): number => {
|
|
299
|
-
roundsSlot.__ameliorateRounds ??= new Map();
|
|
300
|
-
const round = (roundsSlot.__ameliorateRounds.get(planPath) ?? 0) + 1;
|
|
301
|
-
roundsSlot.__ameliorateRounds.set(planPath, round);
|
|
302
|
-
return round;
|
|
303
|
-
};
|
|
304
|
-
|
|
305
|
-
// Lane-level resume (F-007): completed lanes are reused from their
|
|
258
|
+
// Lane-level resume: completed lanes are reused from their
|
|
306
259
|
// persisted outputs; only pending/failed/missing lanes spawn.
|
|
307
260
|
const runnableJobs = jobs.filter((job) => reusable[job.lane.id] === undefined);
|
|
308
261
|
const execution = setupRefinementExecution(
|
|
309
262
|
ctx,
|
|
310
263
|
signal,
|
|
311
|
-
"reviewer",
|
|
312
264
|
runnableJobs.map((job) => ({ id: job.lane.id, label: job.lane.id })),
|
|
313
265
|
modelLabel,
|
|
314
266
|
overlayLang,
|
|
@@ -322,27 +274,16 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
322
274
|
task: job.task,
|
|
323
275
|
cwd: workdir,
|
|
324
276
|
model,
|
|
277
|
+
thinkingLevel: roleConfig.thinking_level ?? undefined,
|
|
325
278
|
tools: subagentTools,
|
|
326
279
|
signal: execution.signal,
|
|
327
280
|
onProgress: (event) => execution.overlay?.update(job.lane.id, event),
|
|
328
281
|
});
|
|
329
282
|
execution.overlay?.complete(job.lane.id, result);
|
|
330
283
|
record(job.name, result.ok ? result.model ?? model : null, result.usage);
|
|
331
|
-
// Persist BEFORE returning
|
|
284
|
+
// Persist BEFORE returning: a crash after this point
|
|
332
285
|
// still leaves the lane reusable.
|
|
333
286
|
persistOutcome(job.lane.id, result.ok ? { ok: true, output: result.output } : { ok: false, error: result.errorMessage });
|
|
334
|
-
if (target === "implementation" && result.ok) {
|
|
335
|
-
try {
|
|
336
|
-
pi.appendEntry("pi-plans-ameliorate", {
|
|
337
|
-
planPath,
|
|
338
|
-
phase: "round",
|
|
339
|
-
currentRound: nextRound(planPath),
|
|
340
|
-
lane: job.lane.id,
|
|
341
|
-
});
|
|
342
|
-
} catch {
|
|
343
|
-
/* appendEntry is best-effort; audit trail survives in subagents.jsonl */
|
|
344
|
-
}
|
|
345
|
-
}
|
|
346
287
|
return { job, result };
|
|
347
288
|
} catch (error) {
|
|
348
289
|
record(job.name, null);
|
|
@@ -385,7 +326,7 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
385
326
|
if (failures === results.length && reusedCount === 0) {
|
|
386
327
|
const first = results[0];
|
|
387
328
|
throw new Error(
|
|
388
|
-
`all reviewer subagents failed: ${first?.result.errorMessage ?? "unknown error"}${first?.result.stderr ? `\nstderr: ${first.result.stderr.slice(0, 2000)}` : ""}${model ? `\nIf the model selector "${model}" is unavailable, reset the confirmation (plans set-role
|
|
329
|
+
`all reviewer subagents failed: ${first?.result.errorMessage ?? "unknown error"}${first?.result.stderr ? `\nstderr: ${first.result.stderr.slice(0, 2000)}` : ""}${model ? `\nIf the model selector "${model}" is unavailable, reset the confirmation (plans set-role, role=reviewer, resetConfirmation: true) — the next refine opens the native model panel to re-confirm.` : ""}`,
|
|
389
330
|
);
|
|
390
331
|
}
|
|
391
332
|
|
|
@@ -398,17 +339,17 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
398
339
|
content: [
|
|
399
340
|
{
|
|
400
341
|
type: "text",
|
|
401
|
-
text: `${text}\n\n---\nConsolidate: merge and dedupe findings into PLAN_vN_reviewer_comments.md${count > 1 ? " (one consolidated file; keep each finding's source reviewer, severity, evidence, and disposition)" : ""}, accept or reject each finding on repo/reference evidence, surface at most five high-priority findings to the user
|
|
342
|
+
text: `${text}\n\n---\nConsolidate: merge and dedupe findings into PLAN_vN_reviewer_comments.md${count > 1 ? " (one consolidated file; keep each finding's source reviewer, severity, evidence, and disposition)" : ""}, accept or reject each finding on repo/reference evidence, merge the Questions sections into one deduped list, surface at most five high-priority findings to the user — then ask EVERY consolidated question with ask_choice (batch them into one questions:[...] form or ask one per call, in the configured language, with stable questionIds), record every answer, and only then revise the plan. After the revision, record the boundary: plans record-checkpoint (checkpoint: { transition: "review-consolidated", roundId: "${roundId}", dispositionArtifact: "<comments file, run-dir relative>" }).`,
|
|
402
343
|
},
|
|
403
344
|
],
|
|
404
345
|
details: {
|
|
405
346
|
mode: "delegated-subagent",
|
|
406
|
-
role: "reviewer",
|
|
407
347
|
planPath,
|
|
408
348
|
roundId,
|
|
409
349
|
reusedLanes: Object.keys(reusable),
|
|
410
350
|
reviewers: count,
|
|
411
351
|
model,
|
|
352
|
+
thinkingLevel: roleConfig.thinking_level,
|
|
412
353
|
outputs: results.map(({ job, result }) => ({ name: job.name, lane: job.lane.id, lens: job.lane.lens, ok: result.ok, output: result.output, stderr: result.stderr, turns: result.turns })),
|
|
413
354
|
},
|
|
414
355
|
};
|
|
@@ -418,14 +359,10 @@ export function registerRefineTool(pi: ExtensionAPI, baseDir: string): void {
|
|
|
418
359
|
},
|
|
419
360
|
|
|
420
361
|
renderCall(args, theme) {
|
|
421
|
-
const count = args.
|
|
422
|
-
let text =
|
|
423
|
-
theme.fg("toolTitle", theme.bold("refine ")) +
|
|
424
|
-
theme.fg("accent", args.role) +
|
|
425
|
-
theme.fg("muted", count > 1 ? ` ×${count}` : "");
|
|
362
|
+
const count = args.reviewers ?? 1;
|
|
363
|
+
let text = theme.fg("toolTitle", theme.bold("refine ")) + theme.fg("accent", "reviewer") + theme.fg("muted", count > 1 ? ` ×${count}` : "");
|
|
426
364
|
const short = args.planPath ? args.planPath.split("/").pop() : "";
|
|
427
365
|
if (short) text += theme.fg("dim", ` ${short}`);
|
|
428
|
-
if (args.target === "implementation") text += theme.fg("dim", " (implementation)");
|
|
429
366
|
if (args.focus) text += `\n${theme.fg("dim", ` focus: ${args.focus.slice(0, 80)}`)}`;
|
|
430
367
|
return new Text(text, 0, 0);
|
|
431
368
|
},
|
package/agents/criticizer.md
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: pi-plans-criticizer
|
|
3
|
-
description: Read-only adversarial questioner for pi-plans refinement rounds; stress-tests plan assumptions with adaptive questions.
|
|
4
|
-
tools: read, grep, find, ls
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
You are a read-only criticizer in the pi-plans workflow.
|
|
8
|
-
|
|
9
|
-
Rules:
|
|
10
|
-
|
|
11
|
-
- Perform read-only analysis. Never edit, write, or delete any file.
|
|
12
|
-
- Stress-test the plan's assumptions; do not rewrite the plan.
|
|
13
|
-
- You may inspect the repository to ground your questions.
|
|
14
|
-
|
|
15
|
-
Output Markdown in exactly this shape:
|
|
16
|
-
|
|
17
|
-
1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
|
|
18
|
-
2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be concrete and answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
|
package/agents/executor.md
DELETED
|
@@ -1,26 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: pi-plans-executor
|
|
3
|
-
description: Delegated plan executor for pi-plans runs; autonomously implements an accepted plan end-to-end in the target workdir and reports Verifier-Checklist progress with markers.
|
|
4
|
-
tools: read, write, edit, bash, grep, find, ls
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
You are a delegated plan executor in the pi-plans workflow. The parent session handed you an accepted plan; you implement it completely and autonomously.
|
|
8
|
-
|
|
9
|
-
Rules:
|
|
10
|
-
|
|
11
|
-
- Work autonomously. You have NO user to ask questions — the ask_choice tool is unavailable in this context. When a decision is genuinely ambiguous, choose the option most consistent with the plan's goals and constraints and record the deviation in your final summary.
|
|
12
|
-
- Implement the plan file you were given, in dependency order. Read the plan first; it is the single source of truth for scope, requirements, and verification steps.
|
|
13
|
-
- Write code directly. Your `write`/`edit` tools operate natively on disk (any DB-first staging in the parent workdir is bypassed for you); no `apply` step is needed.
|
|
14
|
-
- Follow the plan's own execution rules: smallest end-to-end slice first, then layer; no speculative abstractions; no backward-compatibility fallbacks; prefer established libraries already in the project.
|
|
15
|
-
- Emit progress markers IN YOUR REPLIES as you go: `[DONE:VC-xxx]` once a verifier item's stated evidence passes, `[I-###:implemented]` / `[I-###:validating]` for implementation items when the plan defines them. The parent session parses these markers from your streamed messages to update the tracked checklist — put them in message text, not only in the final output.
|
|
16
|
-
- Run the plan's verification steps yourself (tests, validate scripts) and only mark a VC done when its stated evidence actually passes.
|
|
17
|
-
- Do not modify pi-plans state (run.json, checkpoints, ledgers) — the parent owns the run bookkeeping.
|
|
18
|
-
- If a verification step is impossible in this environment, leave the VC unmarked and explain in the summary.
|
|
19
|
-
|
|
20
|
-
Finish with a structured summary in exactly this shape:
|
|
21
|
-
|
|
22
|
-
- Completed VCs: <ids or none>
|
|
23
|
-
- Remaining VCs: <ids or none, with one-line reasons>
|
|
24
|
-
- Implementation items: <per-item state>
|
|
25
|
-
- Deviations from the plan: <any decisions you made on ambiguous points>
|
|
26
|
-
- Evidence: <commands run and their results>
|
|
Binary file
|