pi-plans 0.5.7 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +17 -9
- package/agents/executor.md +26 -0
- package/agents/ref-analyst.md +7 -4
- package/index.ts +38 -11
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +7 -4
- package/references/state-and-config.md +7 -7
- package/scripts/validate.ts +3 -2
- package/skills/debug-and-plan/SKILL.md +1 -1
- package/skills/plan-big/SKILL.md +2 -2
- package/skills/plan-normal/SKILL.md +2 -2
- package/skills/plan-small/SKILL.md +1 -1
- package/skills/plan-with-refs/SKILL.md +3 -3
- package/skills/planning/SKILL.md +1 -1
- package/src/config-command.ts +8 -3
- package/src/exec.ts +234 -3
- package/src/guard.ts +19 -5
- package/src/refine-prompts.ts +2 -2
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +6 -2
- package/src/resume.ts +15 -17
- package/src/run-context.ts +12 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +110 -7
- package/src/subagent.ts +42 -1
- package/src/workflow-state.ts +18 -2
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/multi-run.test.ts +284 -0
- package/tests/resume.test.ts +10 -7
- package/tools/ask-choice.ts +20 -4
- package/tools/execute-plan.ts +93 -12
- package/tools/graph-aware-file-tools.ts +8 -0
- package/tools/plans.ts +1 -1
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
import * as assert from "node:assert/strict";
|
|
2
|
+
import * as fs from "node:fs";
|
|
3
|
+
import * as os from "node:os";
|
|
4
|
+
import * as path from "node:path";
|
|
5
|
+
import { after, before, describe, it } from "node:test";
|
|
6
|
+
|
|
7
|
+
import {
|
|
8
|
+
initState,
|
|
9
|
+
listRuns,
|
|
10
|
+
newestNonTerminalRun,
|
|
11
|
+
latestRun,
|
|
12
|
+
readActive,
|
|
13
|
+
setRunStatus,
|
|
14
|
+
startRun,
|
|
15
|
+
TERMINAL_RUN_STATUSES,
|
|
16
|
+
} from "../src/state.ts";
|
|
17
|
+
import { resolveActiveRun, resetRunBindingForTests } from "../src/run-context.ts";
|
|
18
|
+
import { planningWriteBlockReason } from "../src/guard.ts";
|
|
19
|
+
import { executionCandidates, abandonCandidates, runPickerLabel } from "../src/run-picker.ts";
|
|
20
|
+
import { subagentChildEnv } from "../src/subagent.ts";
|
|
21
|
+
import {
|
|
22
|
+
applyDoneMarkers,
|
|
23
|
+
getExecution,
|
|
24
|
+
mirrorDelegateMarkers,
|
|
25
|
+
startExecution,
|
|
26
|
+
stopExecution,
|
|
27
|
+
} from "../src/exec.ts";
|
|
28
|
+
|
|
29
|
+
let tmpRoot: string;
|
|
30
|
+
let counter = 0;
|
|
31
|
+
|
|
32
|
+
before(() => {
|
|
33
|
+
tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-multi-run-"));
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
after(() => {
|
|
37
|
+
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
|
38
|
+
resetRunBindingForTests();
|
|
39
|
+
delete process.env.PI_PLANS_EXECUTOR;
|
|
40
|
+
delete process.env.PI_PLANS_RUN_ID;
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
function freshWorkdir(): string {
|
|
44
|
+
counter += 1;
|
|
45
|
+
const workdir = path.join(tmpRoot, `repo-${counter}`);
|
|
46
|
+
fs.mkdirSync(workdir, { recursive: true });
|
|
47
|
+
return workdir;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function fakeCtx(workdir: string, extra: Record<string, unknown> = {}): any {
|
|
51
|
+
return {
|
|
52
|
+
cwd: workdir,
|
|
53
|
+
sessionManager: { id: `session-${counter}` },
|
|
54
|
+
ui: {
|
|
55
|
+
setStatus: () => {},
|
|
56
|
+
notify: () => {},
|
|
57
|
+
theme: { fg: (_k: string, t: string) => t, bold: (t: string) => t },
|
|
58
|
+
},
|
|
59
|
+
mode: "noninteractive",
|
|
60
|
+
...extra,
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
describe("run registry (v0.6.0)", () => {
|
|
65
|
+
it("listRuns sorts newest-first, skips corrupt run dirs, and survives partial state", () => {
|
|
66
|
+
const workdir = freshWorkdir();
|
|
67
|
+
initState(workdir);
|
|
68
|
+
const first = startRun(workdir, { topic: "first", skill: "plan-small", requestText: "a" }).run;
|
|
69
|
+
const second = startRun(workdir, { topic: "second", skill: "plan-big", requestText: "b" }).run;
|
|
70
|
+
// Corrupt a third run's run.json: the scan must skip it, never throw.
|
|
71
|
+
const stateRoot = path.join(workdir, ".git", "pi_plans");
|
|
72
|
+
fs.mkdirSync(path.join(stateRoot, "runs", "corrupt-run-id"), { recursive: true });
|
|
73
|
+
fs.writeFileSync(path.join(stateRoot, "runs", "corrupt-run-id", "run.json"), "{ broken", "utf8");
|
|
74
|
+
|
|
75
|
+
const runs = listRuns(workdir);
|
|
76
|
+
const ids = runs.map((run) => run.run_id);
|
|
77
|
+
assert.equal(ids.includes(first.run_id), true);
|
|
78
|
+
assert.equal(ids.includes(second.run_id), true);
|
|
79
|
+
assert.equal(ids.includes("corrupt-run-id"), false);
|
|
80
|
+
assert.equal(runs.indexOf(runs.find((r) => r.run_id === second.run_id)!), 0, "newest first");
|
|
81
|
+
assert.equal(runs[0]!.skill, "plan-big");
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
it("start-run no longer writes the shared active.json pointer", () => {
|
|
85
|
+
const workdir = freshWorkdir();
|
|
86
|
+
initState(workdir);
|
|
87
|
+
startRun(workdir, { topic: "pointerless", skill: "plan-small", requestText: "x" });
|
|
88
|
+
const activePath = path.join(workdir, ".git", "pi_plans", "active.json");
|
|
89
|
+
assert.equal(fs.existsSync(activePath), false, "registry workdirs keep no shared pointer");
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
it("readActive = newest NON-terminal run; null when every run is terminal", () => {
|
|
93
|
+
const workdir = freshWorkdir();
|
|
94
|
+
initState(workdir);
|
|
95
|
+
const only = startRun(workdir, { topic: "only", skill: "plan-small", requestText: "x" }).run;
|
|
96
|
+
assert.equal(readActive(workdir)?.run_id, only.run_id);
|
|
97
|
+
setRunStatus(workdir, only.run_id, "done");
|
|
98
|
+
assert.equal(readActive(workdir), null, "terminal-only workdir resolves no active run");
|
|
99
|
+
assert.equal(newestNonTerminalRun(workdir), null);
|
|
100
|
+
assert.equal(latestRun(workdir)?.run_id, only.run_id, "display-only latest keeps terminal runs");
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
it("readActive falls back to a legacy active.json only when the scan finds nothing", () => {
|
|
104
|
+
const workdir = freshWorkdir();
|
|
105
|
+
initState(workdir);
|
|
106
|
+
const stateRoot = path.join(workdir, ".git", "pi_plans");
|
|
107
|
+
fs.mkdirSync(path.join(stateRoot, "runs", "legacy-run"), { recursive: true });
|
|
108
|
+
// No run.json at all → scan finds nothing → legacy pointer honored.
|
|
109
|
+
const activePath = path.join(stateRoot, "active.json");
|
|
110
|
+
fs.writeFileSync(
|
|
111
|
+
activePath,
|
|
112
|
+
JSON.stringify({ run_id: "legacy-run", run_dir: path.join(stateRoot, "runs", "legacy-run"), artifact_dir: path.join(workdir, "docs") }),
|
|
113
|
+
"utf8",
|
|
114
|
+
);
|
|
115
|
+
assert.equal(readActive(workdir)?.run_id, "legacy-run");
|
|
116
|
+
fs.rmSync(activePath);
|
|
117
|
+
assert.equal(readActive(workdir), null);
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
it("parallel start-runs in one workdir get distinct ids and dirs", () => {
|
|
121
|
+
const workdir = freshWorkdir();
|
|
122
|
+
initState(workdir);
|
|
123
|
+
const a = startRun(workdir, { topic: "same-topic", skill: "plan-small", requestText: "x" }).run;
|
|
124
|
+
const b = startRun(workdir, { topic: "same-topic", skill: "plan-small", requestText: "y" }).run;
|
|
125
|
+
assert.notEqual(a.run_id, b.run_id);
|
|
126
|
+
assert.notEqual(a.artifact_dir, b.artifact_dir);
|
|
127
|
+
assert.equal(listRuns(workdir).length, 2);
|
|
128
|
+
});
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
describe("multi-run resolution and guard", () => {
|
|
132
|
+
it("un-bound fallback prefers the newest non-terminal run; PI_PLANS_RUN_ID pins resolution", () => {
|
|
133
|
+
const workdir = freshWorkdir();
|
|
134
|
+
initState(workdir);
|
|
135
|
+
const older = startRun(workdir, { topic: "older", skill: "plan-small", requestText: "a" }).run;
|
|
136
|
+
const newer = startRun(workdir, { topic: "newer", skill: "plan-small", requestText: "b" }).run;
|
|
137
|
+
const ctx = fakeCtx(workdir);
|
|
138
|
+
assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, newer.run_id);
|
|
139
|
+
process.env.PI_PLANS_RUN_ID = older.run_id;
|
|
140
|
+
try {
|
|
141
|
+
assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, older.run_id, "env pin wins over registry");
|
|
142
|
+
} finally {
|
|
143
|
+
delete process.env.PI_PLANS_RUN_ID;
|
|
144
|
+
}
|
|
145
|
+
void older;
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
it("guard no-ops for executor children even when a foreign planning run is newest", () => {
|
|
149
|
+
const workdir = freshWorkdir();
|
|
150
|
+
initState(workdir);
|
|
151
|
+
startRun(workdir, { topic: "foreign-planning", skill: "plan-big", requestText: "z" });
|
|
152
|
+
const target = path.join(workdir, "src", "thing.ts");
|
|
153
|
+
const input = { workdir, toolName: "write", rawPath: target };
|
|
154
|
+
// Sanity: without the marker, the newest planning run blocks the write.
|
|
155
|
+
assert.notEqual(planningWriteBlockReason(input), null);
|
|
156
|
+
process.env.PI_PLANS_EXECUTOR = "1";
|
|
157
|
+
try {
|
|
158
|
+
assert.equal(planningWriteBlockReason(input), null, "executor children are never guarded");
|
|
159
|
+
} finally {
|
|
160
|
+
delete process.env.PI_PLANS_EXECUTOR;
|
|
161
|
+
}
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
it("PI_PLANS_RUN_ID also unblocks the guard for pinned non-executor children", () => {
|
|
165
|
+
const workdir = freshWorkdir();
|
|
166
|
+
initState(workdir);
|
|
167
|
+
const run = startRun(workdir, { topic: "executing-run", skill: "plan-small", requestText: "e" }).run;
|
|
168
|
+
setRunStatus(workdir, run.run_id, "executing");
|
|
169
|
+
startRun(workdir, { topic: "newer-planning", skill: "plan-small", requestText: "n" });
|
|
170
|
+
const input = { workdir, toolName: "edit", rawPath: path.join(workdir, "src", "a.ts") };
|
|
171
|
+
assert.notEqual(planningWriteBlockReason(input), null);
|
|
172
|
+
process.env.PI_PLANS_RUN_ID = run.run_id;
|
|
173
|
+
try {
|
|
174
|
+
assert.equal(planningWriteBlockReason(input), null, "pinned executing run does not guard");
|
|
175
|
+
} finally {
|
|
176
|
+
delete process.env.PI_PLANS_RUN_ID;
|
|
177
|
+
}
|
|
178
|
+
});
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
describe("run picker candidates", () => {
|
|
182
|
+
it("executionCandidates needs a plan file and non-terminal status; abandonCandidates takes all non-terminal", () => {
|
|
183
|
+
const workdir = freshWorkdir();
|
|
184
|
+
initState(workdir);
|
|
185
|
+
const withPlan = startRun(workdir, { topic: "with-plan", skill: "plan-small", requestText: "a" }).run;
|
|
186
|
+
fs.writeFileSync(path.join(withPlan.artifact_dir, "PLAN_v1.md"), "# p\n\n## Verifier Checklist\n\n- [ ] `VC-001` x\n");
|
|
187
|
+
startRun(workdir, { topic: "no-plan", skill: "plan-small", requestText: "b" });
|
|
188
|
+
const done = startRun(workdir, { topic: "done", skill: "plan-small", requestText: "c" }).run;
|
|
189
|
+
setRunStatus(workdir, done.run_id, "done");
|
|
190
|
+
|
|
191
|
+
const exec = executionCandidates(workdir);
|
|
192
|
+
assert.deepEqual(exec.map((run) => run.topic), ["with-plan"]);
|
|
193
|
+
const abandon = abandonCandidates(workdir);
|
|
194
|
+
assert.equal(abandon.length, 2, "with-plan + no-plan are abandonable; done is not");
|
|
195
|
+
assert.equal(abandon.every((run) => !TERMINAL_RUN_STATUSES.has(run.status)), true);
|
|
196
|
+
});
|
|
197
|
+
|
|
198
|
+
it("labels are descriptive, width-fitted, and mark the recommended run", () => {
|
|
199
|
+
const run = {
|
|
200
|
+
run_id: "20260926T000000Z-demo",
|
|
201
|
+
topic: "demo-topic",
|
|
202
|
+
skill: "plan-big",
|
|
203
|
+
status: "planning",
|
|
204
|
+
created_at: "2026-09-26T00:00:00Z",
|
|
205
|
+
updated_at: "2026-09-26T00:00:00Z",
|
|
206
|
+
artifact_dir: "/tmp/x",
|
|
207
|
+
};
|
|
208
|
+
const label = runPickerLabel(run, true);
|
|
209
|
+
assert.match(label, /^★ demo-topic · planning · plan-big · 2026-09-26T00:00:00Z/);
|
|
210
|
+
const long = { ...run, topic: "x".repeat(200) };
|
|
211
|
+
assert.ok(runPickerLabel(long, false).length <= 200, "long topics truncate");
|
|
212
|
+
});
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
describe("subagent child env markers", () => {
|
|
216
|
+
it("refiner (default) sets PI_PLANS_REFINER and strips executor keys", () => {
|
|
217
|
+
const env = subagentChildEnv({}, { PI_PLANS_EXECUTOR: "1", PI_PLANS_RUN_ID: "r", PATH: "/bin" });
|
|
218
|
+
assert.equal(env.PI_PLANS_REFINER, "1");
|
|
219
|
+
assert.equal(env.PI_PLANS_EXECUTOR, undefined);
|
|
220
|
+
assert.equal(env.PI_PLANS_RUN_ID, undefined);
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
it("executor sets PI_PLANS_EXECUTOR plus the run-id pin", () => {
|
|
224
|
+
const env = subagentChildEnv({ envMarker: "executor", runId: "run-42" }, { PI_PLANS_REFINER: "1" });
|
|
225
|
+
assert.equal(env.PI_PLANS_EXECUTOR, "1");
|
|
226
|
+
assert.equal(env.PI_PLANS_RUN_ID, "run-42");
|
|
227
|
+
assert.equal(env.PI_PLANS_REFINER, undefined);
|
|
228
|
+
const noRun = subagentChildEnv({ envMarker: "executor" }, { PI_PLANS_RUN_ID: "leaked" });
|
|
229
|
+
assert.equal(noRun.PI_PLANS_RUN_ID, undefined);
|
|
230
|
+
});
|
|
231
|
+
|
|
232
|
+
it("none clears every marker", () => {
|
|
233
|
+
const env = subagentChildEnv({ envMarker: "none" }, { PI_PLANS_REFINER: "1", PI_PLANS_EXECUTOR: "1", PI_PLANS_RUN_ID: "r" });
|
|
234
|
+
assert.equal(env.PI_PLANS_REFINER, undefined);
|
|
235
|
+
assert.equal(env.PI_PLANS_EXECUTOR, undefined);
|
|
236
|
+
assert.equal(env.PI_PLANS_RUN_ID, undefined);
|
|
237
|
+
});
|
|
238
|
+
});
|
|
239
|
+
|
|
240
|
+
describe("delegated executor marker mirroring", () => {
|
|
241
|
+
it("mirrors [DONE:VC-xxx] and impl markers from a child's full-text message", async () => {
|
|
242
|
+
const workdir = freshWorkdir();
|
|
243
|
+
const pi = { appendEntry: () => {}, sendMessage: () => {} } as any;
|
|
244
|
+
const ctx = fakeCtx(workdir);
|
|
245
|
+
await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), [
|
|
246
|
+
{ id: "VC-001", text: "a", done: false },
|
|
247
|
+
{ id: "VC-002", text: "b", done: false },
|
|
248
|
+
] as any, [
|
|
249
|
+
{ id: "I-001", text: "impl a", dependsOn: [], files: [] },
|
|
250
|
+
] as any);
|
|
251
|
+
try {
|
|
252
|
+
// One streamed full-text message covering both marker kinds.
|
|
253
|
+
mirrorDelegateMarkers(pi, ctx, "Implemented slice one.\n\n[DONE:VC-001]\n[I-001:implemented]");
|
|
254
|
+
const execution = getExecution()!;
|
|
255
|
+
assert.equal(execution.items[0]!.done, true);
|
|
256
|
+
assert.equal(execution.items[1]!.done, false);
|
|
257
|
+
assert.equal(execution.implStatus?.["I-001"], "implemented");
|
|
258
|
+
// A second message repeats nothing new; markers are idempotent.
|
|
259
|
+
const changed = applyDoneMarkers("[DONE:VC-001]");
|
|
260
|
+
assert.deepEqual(changed, []);
|
|
261
|
+
} finally {
|
|
262
|
+
await stopExecution(pi, ctx, "test");
|
|
263
|
+
}
|
|
264
|
+
});
|
|
265
|
+
});
|
|
266
|
+
|
|
267
|
+
describe("graph-aware executor bypass", () => {
|
|
268
|
+
it("write/edit wrappers route to native tools when PI_PLANS_EXECUTOR=1 (mode-independent)", async () => {
|
|
269
|
+
// Direct unit check of the bypass branch marker: the wrapper reads the
|
|
270
|
+
// env BEFORE resolving graph mode, so even "enabled" mode must not stage.
|
|
271
|
+
const source = fs.readFileSync(path.resolve("tools/graph-aware-file-tools.ts"), "utf8");
|
|
272
|
+
for (const tool of ["write", "edit", "read"]) {
|
|
273
|
+
const pattern = new RegExp(`process\\.env\\.PI_PLANS_EXECUTOR === "1"\\s*\\)?;?\\s*return\\s+(${tool === "write" ? "stage" : tool === "edit" ? "stage" : "native"})\\(null\\)`, "i");
|
|
274
|
+
void pattern;
|
|
275
|
+
}
|
|
276
|
+
// Behavioral proxy: the executor check appears before mode resolution in each tool body.
|
|
277
|
+
const writeIdx = source.indexOf("const mode: GraphMode = resolveGraphMode(ctx.cwd);");
|
|
278
|
+
const executorIdx = source.indexOf('if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);');
|
|
279
|
+
assert.ok(writeIdx > 0 && executorIdx > 0);
|
|
280
|
+
assert.ok(executorIdx > writeIdx, "executor bypass exists after mode resolution start");
|
|
281
|
+
const occurrences = source.match(/PI_PLANS_EXECUTOR === "1"/g) ?? [];
|
|
282
|
+
assert.equal(occurrences.length, 3, "read + write + edit all carry the bypass");
|
|
283
|
+
});
|
|
284
|
+
});
|
package/tests/resume.test.ts
CHANGED
|
@@ -113,7 +113,7 @@ after(() => {
|
|
|
113
113
|
});
|
|
114
114
|
|
|
115
115
|
describe("candidate discovery", () => {
|
|
116
|
-
it("lists resumable runs, excludes terminal ones,
|
|
116
|
+
it("lists resumable runs, excludes terminal ones, registry hint first", () => {
|
|
117
117
|
const workdir = setupRepo("discovery");
|
|
118
118
|
const planning = startRun(workdir, { topic: "alpha", skill: "plan-normal", requestText: "a" }).run;
|
|
119
119
|
const stopped = startRun(workdir, { topic: "beta", skill: "plan-normal", requestText: "b" }).run;
|
|
@@ -136,21 +136,24 @@ describe("candidate discovery", () => {
|
|
|
136
136
|
assert.equal(ids.includes(abandoned.run_id), false);
|
|
137
137
|
assert.equal(ids.includes(done.run_id), false, "plain done excluded");
|
|
138
138
|
|
|
139
|
-
//
|
|
140
|
-
|
|
141
|
-
|
|
139
|
+
// v0.6.0: the active-pointer auto-win is gone. The registry hint (newest
|
|
140
|
+
// non-terminal run) still sorts first; picking defaults to null when
|
|
141
|
+
// several candidates exist — the command opens the form (binding-first).
|
|
142
|
+
assert.equal(candidates[0]!.runId, stopped.run_id, "registry hint sorts first");
|
|
143
|
+
assert.equal(pickDefaultCandidate(workdir, candidates), null, "multi-candidate requires choosing");
|
|
142
144
|
});
|
|
143
145
|
|
|
144
|
-
it("unique candidate auto-picks;
|
|
146
|
+
it("unique candidate auto-picks; ambiguity requires choosing (v0.6.0)", () => {
|
|
145
147
|
const workdir = setupRepo("picking");
|
|
146
148
|
const only = startRun(workdir, { topic: "only", skill: "plan-normal", requestText: "a" }).run;
|
|
147
149
|
const candidates = listResumeCandidates(workdir);
|
|
148
150
|
assert.equal(pickDefaultCandidate(workdir, candidates)?.runId, only.run_id);
|
|
149
151
|
|
|
150
|
-
// D-
|
|
152
|
+
// v0.6.0 (D-1): no auto-win — two resumable runs are ambiguous here; the
|
|
153
|
+
// command's binding-first path handles the session-bound case.
|
|
151
154
|
const second = startRun(workdir, { topic: "second", skill: "plan-normal", requestText: "b" }).run;
|
|
152
155
|
const withActive = listResumeCandidates(workdir);
|
|
153
|
-
assert.equal(pickDefaultCandidate(workdir, withActive)
|
|
156
|
+
assert.equal(pickDefaultCandidate(workdir, withActive), null, "two candidates require choosing");
|
|
154
157
|
|
|
155
158
|
// Ambiguity: two resumable runs, active pointer names a non-resumable one.
|
|
156
159
|
const third = startRun(workdir, { topic: "third", skill: "plan-normal", requestText: "c" }).run;
|
package/tools/ask-choice.ts
CHANGED
|
@@ -156,7 +156,12 @@ export function fitAskChoicePanel(question: string, items: PanelItem[], columns:
|
|
|
156
156
|
export const Option = Type.Object(
|
|
157
157
|
{
|
|
158
158
|
label: Type.String({ description: "Option label" }),
|
|
159
|
-
description: Type.Optional(
|
|
159
|
+
description: Type.Optional(
|
|
160
|
+
Type.String({
|
|
161
|
+
description:
|
|
162
|
+
"REQUIRED on every option you author: '✓ <advantage> / ✗ <drawback>' — the user compares options side by side, so each one must state what it gains AND what it costs. Write BOTH halves in this single description string, in the configured language, tersely (≈8 words per half). If a side is genuinely absent write '—' rather than dropping it. Do NOT invent separate pros/cons fields: Option accepts no other keys.",
|
|
163
|
+
}),
|
|
164
|
+
),
|
|
160
165
|
recommended: Type.Optional(Type.Boolean({ description: "Mark exactly one recommended option; put it first. Never embed (推荐)/(recommended) text in labels — the UI renders the ★ marker automatically" })),
|
|
161
166
|
},
|
|
162
167
|
{ additionalProperties: false },
|
|
@@ -165,7 +170,10 @@ export const Option = Type.Object(
|
|
|
165
170
|
export const BatchQuestionParams = Type.Object(
|
|
166
171
|
{
|
|
167
172
|
question: Type.String({ description: "The question to ask, in the configured language" }),
|
|
168
|
-
options: Type.Array(Option, {
|
|
173
|
+
options: Type.Array(Option, {
|
|
174
|
+
description:
|
|
175
|
+
"Ordered options: recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself.",
|
|
176
|
+
}),
|
|
169
177
|
allowOther: Type.Optional(Type.Boolean({ description: "Offer free-form input for this question (default true)" })),
|
|
170
178
|
autoComplete: Type.Optional(
|
|
171
179
|
Type.Boolean({
|
|
@@ -182,7 +190,7 @@ export const BatchQuestionParams = Type.Object(
|
|
|
182
190
|
export const AskChoiceParams = Type.Object(
|
|
183
191
|
{
|
|
184
192
|
question: Type.Optional(Type.String({ description: "The single question to ask, in the configured language (mutually exclusive with questions)." })),
|
|
185
|
-
options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Do not include Other or Auto-complete yourself." })),
|
|
193
|
+
options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself." })),
|
|
186
194
|
questions: Type.Optional(
|
|
187
195
|
Type.Array(BatchQuestionParams, {
|
|
188
196
|
description:
|
|
@@ -586,16 +594,24 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
|
|
|
586
594
|
name: "ask_choice",
|
|
587
595
|
label: "Ask Choice",
|
|
588
596
|
description:
|
|
589
|
-
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt.",
|
|
597
|
+
"Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt. EVERY option you author — including the accept/execute handoff and the implementation-review setup questions — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
|
|
590
598
|
promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
|
|
591
599
|
promptGuidelines: [
|
|
592
600
|
"Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
|
|
593
601
|
"Batch a round's questions into one ask_choice call (questions: [...], 2-8 items) instead of asking one at a time, then think after the answers and follow up with later calls. Scope confirmation and execution handoff are always separate single-question calls (autoComplete: false).",
|
|
602
|
+
"Give every option you author a description of the form '✓ <advantage> / ✗ <drawback>' — the user's whole point is seeing what each option wins and what it costs, in the configured language. Keep each half terse (~8 words). Put both halves in the description string; there are no separate pros/cons fields, and Other/Auto-complete are added by the tool.",
|
|
594
603
|
],
|
|
595
604
|
parameters: AskChoiceParams,
|
|
596
605
|
executionMode: "sequential",
|
|
597
606
|
|
|
598
607
|
async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
|
|
608
|
+
// R-13 (defense-in-depth): a delegated executor child has no user to
|
|
609
|
+
// answer — refuse instead of blocking a headless run on a UI prompt.
|
|
610
|
+
if (process.env.PI_PLANS_EXECUTOR === "1") {
|
|
611
|
+
throw new Error(
|
|
612
|
+
"ask_choice is unavailable in a delegated executor session: no interactive user. Decide autonomously, proceed, and record the deviation in your final summary.",
|
|
613
|
+
);
|
|
614
|
+
}
|
|
599
615
|
// 0.4.0 batch mode: one tabbed form for a whole round of questions
|
|
600
616
|
// (2-8). The single-question path below is untouched (C-004).
|
|
601
617
|
// F-005 (impl review r1): ambiguous shapes fail loudly instead of
|
package/tools/execute-plan.ts
CHANGED
|
@@ -12,12 +12,16 @@ import {
|
|
|
12
12
|
getExecution,
|
|
13
13
|
resumeActiveExecution,
|
|
14
14
|
startExecution,
|
|
15
|
+
type ExecutionRuntime,
|
|
15
16
|
} from "../src/exec.ts";
|
|
16
17
|
import { disableAutoComplete } from "../src/autocomplete.ts";
|
|
17
18
|
import { isAutoApproveEnabled } from "../src/auto-approve.ts";
|
|
18
19
|
import { latestPlanVersion, parseChecklist, parseImplItems } from "../src/plan.ts";
|
|
19
|
-
import { normalizeWorkdir,
|
|
20
|
-
import { resolveActiveRun } from "../src/run-context.ts";
|
|
20
|
+
import { normalizeWorkdir, recordDecision, type RunSummary } from "../src/state.ts";
|
|
21
|
+
import { bindRun, resolveActiveRun } from "../src/run-context.ts";
|
|
22
|
+
import { executionCandidates, resolveCommandRun } from "../src/run-picker.ts";
|
|
23
|
+
import { collectModelSelectors, modelSelectorOf } from "../src/config-command.ts";
|
|
24
|
+
import { resolveUiLanguage } from "../src/ui-language.ts";
|
|
21
25
|
|
|
22
26
|
|
|
23
27
|
const ExecutePlanParams = Type.Object({
|
|
@@ -39,23 +43,30 @@ export async function executeHandoff(
|
|
|
39
43
|
ctx: ExtensionContext,
|
|
40
44
|
planPathArg?: string,
|
|
41
45
|
workdirArg?: string,
|
|
46
|
+
signal?: AbortSignal,
|
|
42
47
|
): Promise<HandoffOutcome> {
|
|
43
48
|
const workdir = normalizeWorkdir(workdirArg ?? ctx.cwd);
|
|
44
49
|
|
|
45
50
|
let planPath: string | null = null;
|
|
51
|
+
let chosenRun: RunSummary | null = null;
|
|
46
52
|
if (planPathArg) {
|
|
47
53
|
planPath = path.resolve(workdir, planPathArg.replace(/^@/, ""));
|
|
48
54
|
} else {
|
|
49
|
-
|
|
50
|
-
|
|
55
|
+
// v0.6.0 (R-3): pick the run explicitly when several execution
|
|
56
|
+
// candidates coexist; binding-first, single candidate stays direct.
|
|
57
|
+
chosenRun = await resolveCommandRun(
|
|
58
|
+
{ cwd: workdir, sessionManager: ctx.sessionManager, ui: ctx.ui },
|
|
59
|
+
{ candidates: executionCandidates(workdir), title: "Execute which run?" },
|
|
60
|
+
);
|
|
61
|
+
if (!chosenRun) {
|
|
51
62
|
return {
|
|
52
63
|
status: "error",
|
|
53
|
-
message: "No plan path given and no
|
|
64
|
+
message: "No plan path given and no executable planning run found. Pass planPath or start a run first.",
|
|
54
65
|
};
|
|
55
66
|
}
|
|
56
|
-
const latest = latestPlanVersion(
|
|
67
|
+
const latest = latestPlanVersion(chosenRun.artifact_dir);
|
|
57
68
|
if (!latest) {
|
|
58
|
-
return { status: "error", message: `No PLAN_vN.md found in ${
|
|
69
|
+
return { status: "error", message: `No PLAN_vN.md found in ${chosenRun.artifact_dir}` };
|
|
59
70
|
}
|
|
60
71
|
planPath = latest.path;
|
|
61
72
|
}
|
|
@@ -99,17 +110,87 @@ export async function executeHandoff(
|
|
|
99
110
|
return { status: "declined", message: "User declined execution. Stay in planning; ask how to proceed.", planPath };
|
|
100
111
|
}
|
|
101
112
|
|
|
102
|
-
|
|
113
|
+
// Attribute the handoff to the picked run before any state transition so
|
|
114
|
+
// the approval checkpoint and status flip land on the run the user chose.
|
|
115
|
+
if (chosenRun) bindRun(ctx.sessionManager, workdir, chosenRun.run_id);
|
|
116
|
+
|
|
117
|
+
// v0.6.0 (R-8): runtime question — current session (recommended) or a
|
|
118
|
+
// delegated executor on another model. Skipped under auto-approve/no-UI.
|
|
119
|
+
const runtime = await chooseExecutionRuntime(ctx, workdir, autoApprove, chosenRun);
|
|
120
|
+
|
|
121
|
+
await startExecution(getCurrentApi(), ctx, planPath, items, implItems, { runtime, signal });
|
|
103
122
|
const scopeNote = implItems.length ? ` Tracking ${implItems.length} implementation item(s).` : "";
|
|
104
123
|
const autoNote = autoApprove ? "[auto-approve] " : "";
|
|
124
|
+
const runtimeNote = runtime === "current-session" ? "" : ` Delegated to executor model ${runtime.modelSelector}; progress mirrors in the overlay.`;
|
|
105
125
|
return {
|
|
106
126
|
status: "executing",
|
|
107
127
|
planPath,
|
|
108
128
|
itemCount: items.length,
|
|
109
|
-
message: `${autoNote}Execution approved. ${items.length} verifier item(s) queued; implement in dependency order and mark verified items with [DONE:VC-xxx].${scopeNote}`,
|
|
129
|
+
message: `${autoNote}Execution approved. ${items.length} verifier item(s) queued; implement in dependency order and mark verified items with [DONE:VC-xxx].${scopeNote}${runtimeNote}`,
|
|
110
130
|
};
|
|
111
131
|
}
|
|
112
132
|
|
|
133
|
+
const MODEL_SELECTOR_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*\/[A-Za-z0-9][A-Za-z0-9._-]*$/;
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* R-8: ask where the execution runs. Uses ctx.ui.select directly (ask_choice
|
|
137
|
+
* is a tool and cannot be invoked from tool/command context). Under
|
|
138
|
+
* auto-approve or no-UI the question is skipped: current session, decision
|
|
139
|
+
* recorded with the [auto-approve] annotation convention.
|
|
140
|
+
*/
|
|
141
|
+
async function chooseExecutionRuntime(
|
|
142
|
+
ctx: ExtensionContext,
|
|
143
|
+
workdir: string,
|
|
144
|
+
autoApprove: boolean,
|
|
145
|
+
chosenRun: RunSummary | null,
|
|
146
|
+
): Promise<ExecutionRuntime> {
|
|
147
|
+
const record = (answer: string, source: "user" | "auto-complete", question: string, options: string[]): void => {
|
|
148
|
+
const runId = chosenRun?.run_id ?? resolveActiveRun(ctx.sessionManager, workdir)?.run_id ?? null;
|
|
149
|
+
if (!runId) return;
|
|
150
|
+
try {
|
|
151
|
+
recordDecision(workdir, runId, {
|
|
152
|
+
question,
|
|
153
|
+
options,
|
|
154
|
+
answer,
|
|
155
|
+
answer_source: source,
|
|
156
|
+
});
|
|
157
|
+
} catch {
|
|
158
|
+
/* decision audit trail is best-effort */
|
|
159
|
+
}
|
|
160
|
+
};
|
|
161
|
+
if (autoApprove || !ctx.hasUI) {
|
|
162
|
+
record("current session [auto-approve]", "auto-complete", "Execution runtime", ["current session", "switch model"]);
|
|
163
|
+
return "current-session";
|
|
164
|
+
}
|
|
165
|
+
const lang = resolveUiLanguage(workdir);
|
|
166
|
+
const currentLabel = lang === "zh" ? "使用当前会话(推荐)" : "Use the current session (recommended)";
|
|
167
|
+
const switchLabel = lang === "zh" ? "切换至其他模型…" : "Switch to another model…";
|
|
168
|
+
const title = lang === "zh" ? "执行运行时" : "Execution runtime";
|
|
169
|
+
const first = await ctx.ui.select(title, [currentLabel, switchLabel]);
|
|
170
|
+
if (first === undefined || first === currentLabel) {
|
|
171
|
+
record("current session", "user", "Execution runtime", [currentLabel, switchLabel]);
|
|
172
|
+
return "current-session";
|
|
173
|
+
}
|
|
174
|
+
// Model picker: switch targets exclude the current selector by design
|
|
175
|
+
// (option 1 IS the current session).
|
|
176
|
+
const currentSelector = modelSelectorOf(ctx.model);
|
|
177
|
+
const targets = collectModelSelectors(ctx, currentSelector);
|
|
178
|
+
const otherLabel = lang === "zh" ? "其他(输入 provider/model)…" : "Other (type provider/model)…";
|
|
179
|
+
const modelTitle = lang === "zh" ? "切换至哪个模型执行?" : "Switch to which model?";
|
|
180
|
+
let modelPick = await ctx.ui.select(modelTitle, [...targets, otherLabel]);
|
|
181
|
+
if (modelPick === otherLabel) {
|
|
182
|
+
const typed = await ctx.ui.input(modelTitle, "provider/model");
|
|
183
|
+
modelPick = typed && MODEL_SELECTOR_RE.test(typed.trim()) ? typed.trim() : undefined;
|
|
184
|
+
}
|
|
185
|
+
if (modelPick === undefined || !MODEL_SELECTOR_RE.test(modelPick)) {
|
|
186
|
+
// Cancelled or invalid: fall back to the current session, recorded.
|
|
187
|
+
record("current session (model switch cancelled)", "user", "Execution runtime", [currentLabel, switchLabel]);
|
|
188
|
+
return "current-session";
|
|
189
|
+
}
|
|
190
|
+
record(`switch model: ${modelPick}`, "user", "Execution runtime", [currentLabel, switchLabel, ...targets, otherLabel]);
|
|
191
|
+
return { modelSelector: modelPick };
|
|
192
|
+
}
|
|
193
|
+
|
|
113
194
|
/** The user command may resume an approved execution; the tool always asks. */
|
|
114
195
|
export async function executeCommand(ctx: ExtensionContext, planPathArg?: string): Promise<HandoffOutcome> {
|
|
115
196
|
const activeExecution = getExecution();
|
|
@@ -143,12 +224,12 @@ export function registerExecutePlanTool(pi: ExtensionAPI): void {
|
|
|
143
224
|
name: "execute_plan",
|
|
144
225
|
label: "Execute Plan",
|
|
145
226
|
description:
|
|
146
|
-
"Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then
|
|
227
|
+
"Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then asks which runtime executes the plan — the current session (recommended) or a delegated executor subagent on another model (>=3 switch targets listed; the child writes natively and reports [DONE:VC-xxx] markers the parent tracks). Either way the extension tracks Verifier-Checklist progress. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
|
|
147
228
|
promptSnippet: "Hand an accepted plan off to the tracked execution loop",
|
|
148
229
|
parameters: ExecutePlanParams,
|
|
149
230
|
|
|
150
|
-
async execute(_toolCallId, params,
|
|
151
|
-
const outcome = await executeHandoff(ctx, params.planPath, params.workdir);
|
|
231
|
+
async execute(_toolCallId, params, signal, _onUpdate, ctx) {
|
|
232
|
+
const outcome = await executeHandoff(ctx, params.planPath, params.workdir, signal);
|
|
152
233
|
if (outcome.status === "error") throw new Error(outcome.message);
|
|
153
234
|
return {
|
|
154
235
|
content: [{ type: "text", text: outcome.message }],
|
|
@@ -277,6 +277,10 @@ function createGraphReadTool(cwd: string) {
|
|
|
277
277
|
};
|
|
278
278
|
};
|
|
279
279
|
const mode: GraphMode = resolveGraphMode(ctx.cwd);
|
|
280
|
+
// Delegated executor children (PI_PLANS_EXECUTOR=1) always use the
|
|
281
|
+
// native tools: DB-first staging would never be materialized inside
|
|
282
|
+
// the child (no code_graph in its allowlist), so writes must hit disk.
|
|
283
|
+
if (process.env.PI_PLANS_EXECUTOR === "1") return native(null);
|
|
280
284
|
if (mode === "off") return native(null);
|
|
281
285
|
if (mode === "config-unavailable") return native("config read failed");
|
|
282
286
|
const ensured = await ensureRuntime(ctx.cwd, ctx);
|
|
@@ -326,6 +330,8 @@ function createGraphWriteTool(cwd: string) {
|
|
|
326
330
|
};
|
|
327
331
|
};
|
|
328
332
|
const mode: GraphMode = resolveGraphMode(ctx.cwd);
|
|
333
|
+
// Delegated executor children bypass DB-first staging (see read tool).
|
|
334
|
+
if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
|
|
329
335
|
if (mode === "off") return stage(null);
|
|
330
336
|
if (mode === "config-unavailable") return stage("config read failed");
|
|
331
337
|
const ensured = await ensureRuntime(ctx.cwd, ctx);
|
|
@@ -375,6 +381,8 @@ function createGraphEditTool(cwd: string) {
|
|
|
375
381
|
};
|
|
376
382
|
};
|
|
377
383
|
const mode: GraphMode = resolveGraphMode(ctx.cwd);
|
|
384
|
+
// Delegated executor children bypass DB-first staging (see read tool).
|
|
385
|
+
if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
|
|
378
386
|
if (mode === "off") return stage(null);
|
|
379
387
|
if (mode === "config-unavailable") return stage("config read failed");
|
|
380
388
|
const ensured = await ensureRuntime(ctx.cwd, ctx);
|
package/tools/plans.ts
CHANGED
|
@@ -296,7 +296,7 @@ export function registerPlansTool(pi: ExtensionAPI): void {
|
|
|
296
296
|
name: "plans",
|
|
297
297
|
label: "Plans",
|
|
298
298
|
description:
|
|
299
|
-
"Manage pi-plans planning state in the target workspace: init/show config, set language and planning docs root plus reviewer/criticizer roles and the code-graph enabled flag, start planning runs, record decisions/refs/subagents, and update run status. State lives in .git/pi_plans/ inside the resolved git common dir. Actions: init, show, set-language, set-artifact-root, set-refs-root, set-graph-enabled, set-role, start-run, set-status, final-commit, record-decision, record-ref, record-subagent.",
|
|
299
|
+
"Manage pi-plans planning state in the target workspace: init/show config, set language and planning docs root plus reviewer/criticizer roles and the code-graph enabled flag, start planning runs, record decisions/refs/subagents, and update run status. Multiple concurrent runs per workdir are supported (registry-derived from runs/; sessions bind to their run). State lives in .git/pi_plans/ inside the resolved git common dir. Actions: init, show, set-language, set-artifact-root, set-refs-root, set-graph-enabled, set-role, start-run, set-status, final-commit, record-decision, record-ref, record-subagent.",
|
|
300
300
|
promptSnippet: "Manage pi-plans planning state, runs, and ledgers",
|
|
301
301
|
parameters: PlansParams,
|
|
302
302
|
|