pi-plans 0.5.7 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,284 @@
1
+ import * as assert from "node:assert/strict";
2
+ import * as fs from "node:fs";
3
+ import * as os from "node:os";
4
+ import * as path from "node:path";
5
+ import { after, before, describe, it } from "node:test";
6
+
7
+ import {
8
+ initState,
9
+ listRuns,
10
+ newestNonTerminalRun,
11
+ latestRun,
12
+ readActive,
13
+ setRunStatus,
14
+ startRun,
15
+ TERMINAL_RUN_STATUSES,
16
+ } from "../src/state.ts";
17
+ import { resolveActiveRun, resetRunBindingForTests } from "../src/run-context.ts";
18
+ import { planningWriteBlockReason } from "../src/guard.ts";
19
+ import { executionCandidates, abandonCandidates, runPickerLabel } from "../src/run-picker.ts";
20
+ import { subagentChildEnv } from "../src/subagent.ts";
21
+ import {
22
+ applyDoneMarkers,
23
+ getExecution,
24
+ mirrorDelegateMarkers,
25
+ startExecution,
26
+ stopExecution,
27
+ } from "../src/exec.ts";
28
+
29
+ let tmpRoot: string;
30
+ let counter = 0;
31
+
32
+ before(() => {
33
+ tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-multi-run-"));
34
+ });
35
+
36
+ after(() => {
37
+ fs.rmSync(tmpRoot, { recursive: true, force: true });
38
+ resetRunBindingForTests();
39
+ delete process.env.PI_PLANS_EXECUTOR;
40
+ delete process.env.PI_PLANS_RUN_ID;
41
+ });
42
+
43
+ function freshWorkdir(): string {
44
+ counter += 1;
45
+ const workdir = path.join(tmpRoot, `repo-${counter}`);
46
+ fs.mkdirSync(workdir, { recursive: true });
47
+ return workdir;
48
+ }
49
+
50
+ function fakeCtx(workdir: string, extra: Record<string, unknown> = {}): any {
51
+ return {
52
+ cwd: workdir,
53
+ sessionManager: { id: `session-${counter}` },
54
+ ui: {
55
+ setStatus: () => {},
56
+ notify: () => {},
57
+ theme: { fg: (_k: string, t: string) => t, bold: (t: string) => t },
58
+ },
59
+ mode: "noninteractive",
60
+ ...extra,
61
+ };
62
+ }
63
+
64
+ describe("run registry (v0.6.0)", () => {
65
+ it("listRuns sorts newest-first, skips corrupt run dirs, and survives partial state", () => {
66
+ const workdir = freshWorkdir();
67
+ initState(workdir);
68
+ const first = startRun(workdir, { topic: "first", skill: "plan-small", requestText: "a" }).run;
69
+ const second = startRun(workdir, { topic: "second", skill: "plan-big", requestText: "b" }).run;
70
+ // Corrupt a third run's run.json: the scan must skip it, never throw.
71
+ const stateRoot = path.join(workdir, ".git", "pi_plans");
72
+ fs.mkdirSync(path.join(stateRoot, "runs", "corrupt-run-id"), { recursive: true });
73
+ fs.writeFileSync(path.join(stateRoot, "runs", "corrupt-run-id", "run.json"), "{ broken", "utf8");
74
+
75
+ const runs = listRuns(workdir);
76
+ const ids = runs.map((run) => run.run_id);
77
+ assert.equal(ids.includes(first.run_id), true);
78
+ assert.equal(ids.includes(second.run_id), true);
79
+ assert.equal(ids.includes("corrupt-run-id"), false);
80
+ assert.equal(runs.indexOf(runs.find((r) => r.run_id === second.run_id)!), 0, "newest first");
81
+ assert.equal(runs[0]!.skill, "plan-big");
82
+ });
83
+
84
+ it("start-run no longer writes the shared active.json pointer", () => {
85
+ const workdir = freshWorkdir();
86
+ initState(workdir);
87
+ startRun(workdir, { topic: "pointerless", skill: "plan-small", requestText: "x" });
88
+ const activePath = path.join(workdir, ".git", "pi_plans", "active.json");
89
+ assert.equal(fs.existsSync(activePath), false, "registry workdirs keep no shared pointer");
90
+ });
91
+
92
+ it("readActive = newest NON-terminal run; null when every run is terminal", () => {
93
+ const workdir = freshWorkdir();
94
+ initState(workdir);
95
+ const only = startRun(workdir, { topic: "only", skill: "plan-small", requestText: "x" }).run;
96
+ assert.equal(readActive(workdir)?.run_id, only.run_id);
97
+ setRunStatus(workdir, only.run_id, "done");
98
+ assert.equal(readActive(workdir), null, "terminal-only workdir resolves no active run");
99
+ assert.equal(newestNonTerminalRun(workdir), null);
100
+ assert.equal(latestRun(workdir)?.run_id, only.run_id, "display-only latest keeps terminal runs");
101
+ });
102
+
103
+ it("readActive falls back to a legacy active.json only when the scan finds nothing", () => {
104
+ const workdir = freshWorkdir();
105
+ initState(workdir);
106
+ const stateRoot = path.join(workdir, ".git", "pi_plans");
107
+ fs.mkdirSync(path.join(stateRoot, "runs", "legacy-run"), { recursive: true });
108
+ // No run.json at all → scan finds nothing → legacy pointer honored.
109
+ const activePath = path.join(stateRoot, "active.json");
110
+ fs.writeFileSync(
111
+ activePath,
112
+ JSON.stringify({ run_id: "legacy-run", run_dir: path.join(stateRoot, "runs", "legacy-run"), artifact_dir: path.join(workdir, "docs") }),
113
+ "utf8",
114
+ );
115
+ assert.equal(readActive(workdir)?.run_id, "legacy-run");
116
+ fs.rmSync(activePath);
117
+ assert.equal(readActive(workdir), null);
118
+ });
119
+
120
+ it("parallel start-runs in one workdir get distinct ids and dirs", () => {
121
+ const workdir = freshWorkdir();
122
+ initState(workdir);
123
+ const a = startRun(workdir, { topic: "same-topic", skill: "plan-small", requestText: "x" }).run;
124
+ const b = startRun(workdir, { topic: "same-topic", skill: "plan-small", requestText: "y" }).run;
125
+ assert.notEqual(a.run_id, b.run_id);
126
+ assert.notEqual(a.artifact_dir, b.artifact_dir);
127
+ assert.equal(listRuns(workdir).length, 2);
128
+ });
129
+ });
130
+
131
+ describe("multi-run resolution and guard", () => {
132
+ it("un-bound fallback prefers the newest non-terminal run; PI_PLANS_RUN_ID pins resolution", () => {
133
+ const workdir = freshWorkdir();
134
+ initState(workdir);
135
+ const older = startRun(workdir, { topic: "older", skill: "plan-small", requestText: "a" }).run;
136
+ const newer = startRun(workdir, { topic: "newer", skill: "plan-small", requestText: "b" }).run;
137
+ const ctx = fakeCtx(workdir);
138
+ assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, newer.run_id);
139
+ process.env.PI_PLANS_RUN_ID = older.run_id;
140
+ try {
141
+ assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, older.run_id, "env pin wins over registry");
142
+ } finally {
143
+ delete process.env.PI_PLANS_RUN_ID;
144
+ }
145
+ void older;
146
+ });
147
+
148
+ it("guard no-ops for executor children even when a foreign planning run is newest", () => {
149
+ const workdir = freshWorkdir();
150
+ initState(workdir);
151
+ startRun(workdir, { topic: "foreign-planning", skill: "plan-big", requestText: "z" });
152
+ const target = path.join(workdir, "src", "thing.ts");
153
+ const input = { workdir, toolName: "write", rawPath: target };
154
+ // Sanity: without the marker, the newest planning run blocks the write.
155
+ assert.notEqual(planningWriteBlockReason(input), null);
156
+ process.env.PI_PLANS_EXECUTOR = "1";
157
+ try {
158
+ assert.equal(planningWriteBlockReason(input), null, "executor children are never guarded");
159
+ } finally {
160
+ delete process.env.PI_PLANS_EXECUTOR;
161
+ }
162
+ });
163
+
164
+ it("PI_PLANS_RUN_ID also unblocks the guard for pinned non-executor children", () => {
165
+ const workdir = freshWorkdir();
166
+ initState(workdir);
167
+ const run = startRun(workdir, { topic: "executing-run", skill: "plan-small", requestText: "e" }).run;
168
+ setRunStatus(workdir, run.run_id, "executing");
169
+ startRun(workdir, { topic: "newer-planning", skill: "plan-small", requestText: "n" });
170
+ const input = { workdir, toolName: "edit", rawPath: path.join(workdir, "src", "a.ts") };
171
+ assert.notEqual(planningWriteBlockReason(input), null);
172
+ process.env.PI_PLANS_RUN_ID = run.run_id;
173
+ try {
174
+ assert.equal(planningWriteBlockReason(input), null, "pinned executing run does not guard");
175
+ } finally {
176
+ delete process.env.PI_PLANS_RUN_ID;
177
+ }
178
+ });
179
+ });
180
+
181
+ describe("run picker candidates", () => {
182
+ it("executionCandidates needs a plan file and non-terminal status; abandonCandidates takes all non-terminal", () => {
183
+ const workdir = freshWorkdir();
184
+ initState(workdir);
185
+ const withPlan = startRun(workdir, { topic: "with-plan", skill: "plan-small", requestText: "a" }).run;
186
+ fs.writeFileSync(path.join(withPlan.artifact_dir, "PLAN_v1.md"), "# p\n\n## Verifier Checklist\n\n- [ ] `VC-001` x\n");
187
+ startRun(workdir, { topic: "no-plan", skill: "plan-small", requestText: "b" });
188
+ const done = startRun(workdir, { topic: "done", skill: "plan-small", requestText: "c" }).run;
189
+ setRunStatus(workdir, done.run_id, "done");
190
+
191
+ const exec = executionCandidates(workdir);
192
+ assert.deepEqual(exec.map((run) => run.topic), ["with-plan"]);
193
+ const abandon = abandonCandidates(workdir);
194
+ assert.equal(abandon.length, 2, "with-plan + no-plan are abandonable; done is not");
195
+ assert.equal(abandon.every((run) => !TERMINAL_RUN_STATUSES.has(run.status)), true);
196
+ });
197
+
198
+ it("labels are descriptive, width-fitted, and mark the recommended run", () => {
199
+ const run = {
200
+ run_id: "20260926T000000Z-demo",
201
+ topic: "demo-topic",
202
+ skill: "plan-big",
203
+ status: "planning",
204
+ created_at: "2026-09-26T00:00:00Z",
205
+ updated_at: "2026-09-26T00:00:00Z",
206
+ artifact_dir: "/tmp/x",
207
+ };
208
+ const label = runPickerLabel(run, true);
209
+ assert.match(label, /^★ demo-topic · planning · plan-big · 2026-09-26T00:00:00Z/);
210
+ const long = { ...run, topic: "x".repeat(200) };
211
+ assert.ok(runPickerLabel(long, false).length <= 200, "long topics truncate");
212
+ });
213
+ });
214
+
215
+ describe("subagent child env markers", () => {
216
+ it("refiner (default) sets PI_PLANS_REFINER and strips executor keys", () => {
217
+ const env = subagentChildEnv({}, { PI_PLANS_EXECUTOR: "1", PI_PLANS_RUN_ID: "r", PATH: "/bin" });
218
+ assert.equal(env.PI_PLANS_REFINER, "1");
219
+ assert.equal(env.PI_PLANS_EXECUTOR, undefined);
220
+ assert.equal(env.PI_PLANS_RUN_ID, undefined);
221
+ });
222
+
223
+ it("executor sets PI_PLANS_EXECUTOR plus the run-id pin", () => {
224
+ const env = subagentChildEnv({ envMarker: "executor", runId: "run-42" }, { PI_PLANS_REFINER: "1" });
225
+ assert.equal(env.PI_PLANS_EXECUTOR, "1");
226
+ assert.equal(env.PI_PLANS_RUN_ID, "run-42");
227
+ assert.equal(env.PI_PLANS_REFINER, undefined);
228
+ const noRun = subagentChildEnv({ envMarker: "executor" }, { PI_PLANS_RUN_ID: "leaked" });
229
+ assert.equal(noRun.PI_PLANS_RUN_ID, undefined);
230
+ });
231
+
232
+ it("none clears every marker", () => {
233
+ const env = subagentChildEnv({ envMarker: "none" }, { PI_PLANS_REFINER: "1", PI_PLANS_EXECUTOR: "1", PI_PLANS_RUN_ID: "r" });
234
+ assert.equal(env.PI_PLANS_REFINER, undefined);
235
+ assert.equal(env.PI_PLANS_EXECUTOR, undefined);
236
+ assert.equal(env.PI_PLANS_RUN_ID, undefined);
237
+ });
238
+ });
239
+
240
+ describe("delegated executor marker mirroring", () => {
241
+ it("mirrors [DONE:VC-xxx] and impl markers from a child's full-text message", async () => {
242
+ const workdir = freshWorkdir();
243
+ const pi = { appendEntry: () => {}, sendMessage: () => {} } as any;
244
+ const ctx = fakeCtx(workdir);
245
+ await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), [
246
+ { id: "VC-001", text: "a", done: false },
247
+ { id: "VC-002", text: "b", done: false },
248
+ ] as any, [
249
+ { id: "I-001", text: "impl a", dependsOn: [], files: [] },
250
+ ] as any);
251
+ try {
252
+ // One streamed full-text message covering both marker kinds.
253
+ mirrorDelegateMarkers(pi, ctx, "Implemented slice one.\n\n[DONE:VC-001]\n[I-001:implemented]");
254
+ const execution = getExecution()!;
255
+ assert.equal(execution.items[0]!.done, true);
256
+ assert.equal(execution.items[1]!.done, false);
257
+ assert.equal(execution.implStatus?.["I-001"], "implemented");
258
+ // A second message repeats nothing new; markers are idempotent.
259
+ const changed = applyDoneMarkers("[DONE:VC-001]");
260
+ assert.deepEqual(changed, []);
261
+ } finally {
262
+ await stopExecution(pi, ctx, "test");
263
+ }
264
+ });
265
+ });
266
+
267
+ describe("graph-aware executor bypass", () => {
268
+ it("write/edit wrappers route to native tools when PI_PLANS_EXECUTOR=1 (mode-independent)", async () => {
269
+ // Direct unit check of the bypass branch marker: the wrapper reads the
270
+ // env BEFORE resolving graph mode, so even "enabled" mode must not stage.
271
+ const source = fs.readFileSync(path.resolve("tools/graph-aware-file-tools.ts"), "utf8");
272
+ for (const tool of ["write", "edit", "read"]) {
273
+ const pattern = new RegExp(`process\\.env\\.PI_PLANS_EXECUTOR === "1"\\s*\\)?;?\\s*return\\s+(${tool === "write" ? "stage" : tool === "edit" ? "stage" : "native"})\\(null\\)`, "i");
274
+ void pattern;
275
+ }
276
+ // Behavioral proxy: the executor check appears before mode resolution in each tool body.
277
+ const writeIdx = source.indexOf("const mode: GraphMode = resolveGraphMode(ctx.cwd);");
278
+ const executorIdx = source.indexOf('if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);');
279
+ assert.ok(writeIdx > 0 && executorIdx > 0);
280
+ assert.ok(executorIdx > writeIdx, "executor bypass exists after mode resolution start");
281
+ const occurrences = source.match(/PI_PLANS_EXECUTOR === "1"/g) ?? [];
282
+ assert.equal(occurrences.length, 3, "read + write + edit all carry the bypass");
283
+ });
284
+ });
@@ -113,7 +113,7 @@ after(() => {
113
113
  });
114
114
 
115
115
  describe("candidate discovery", () => {
116
- it("lists resumable runs, excludes terminal ones, prioritizes active", () => {
116
+ it("lists resumable runs, excludes terminal ones, registry hint first", () => {
117
117
  const workdir = setupRepo("discovery");
118
118
  const planning = startRun(workdir, { topic: "alpha", skill: "plan-normal", requestText: "a" }).run;
119
119
  const stopped = startRun(workdir, { topic: "beta", skill: "plan-normal", requestText: "b" }).run;
@@ -136,21 +136,24 @@ describe("candidate discovery", () => {
136
136
  assert.equal(ids.includes(abandoned.run_id), false);
137
137
  assert.equal(ids.includes(done.run_id), false, "plain done excluded");
138
138
 
139
- // Active priority: shared pointer names `doneWithReview` (last started).
140
- const picked = pickDefaultCandidate(workdir, candidates);
141
- assert.equal(picked?.runId, doneWithReview.run_id);
139
+ // v0.6.0: the active-pointer auto-win is gone. The registry hint (newest
140
+ // non-terminal run) still sorts first; picking defaults to null when
141
+ // several candidates exist — the command opens the form (binding-first).
142
+ assert.equal(candidates[0]!.runId, stopped.run_id, "registry hint sorts first");
143
+ assert.equal(pickDefaultCandidate(workdir, candidates), null, "multi-candidate requires choosing");
142
144
  });
143
145
 
144
- it("unique candidate auto-picks; unfinished active wins; ambiguity requires choosing", () => {
146
+ it("unique candidate auto-picks; ambiguity requires choosing (v0.6.0)", () => {
145
147
  const workdir = setupRepo("picking");
146
148
  const only = startRun(workdir, { topic: "only", skill: "plan-normal", requestText: "a" }).run;
147
149
  const candidates = listResumeCandidates(workdir);
148
150
  assert.equal(pickDefaultCandidate(workdir, candidates)?.runId, only.run_id);
149
151
 
150
- // D-001: an unfinished active run wins even when others exist.
152
+ // v0.6.0 (D-1): no auto-win — two resumable runs are ambiguous here; the
153
+ // command's binding-first path handles the session-bound case.
151
154
  const second = startRun(workdir, { topic: "second", skill: "plan-normal", requestText: "b" }).run;
152
155
  const withActive = listResumeCandidates(workdir);
153
- assert.equal(pickDefaultCandidate(workdir, withActive)?.runId, second.run_id);
156
+ assert.equal(pickDefaultCandidate(workdir, withActive), null, "two candidates require choosing");
154
157
 
155
158
  // Ambiguity: two resumable runs, active pointer names a non-resumable one.
156
159
  const third = startRun(workdir, { topic: "third", skill: "plan-normal", requestText: "c" }).run;
@@ -156,7 +156,12 @@ export function fitAskChoicePanel(question: string, items: PanelItem[], columns:
156
156
  export const Option = Type.Object(
157
157
  {
158
158
  label: Type.String({ description: "Option label" }),
159
- description: Type.Optional(Type.String({ description: "Short tradeoff that matters, shown to the user" })),
159
+ description: Type.Optional(
160
+ Type.String({
161
+ description:
162
+ "REQUIRED on every option you author: '✓ <advantage> / ✗ <drawback>' — the user compares options side by side, so each one must state what it gains AND what it costs. Write BOTH halves in this single description string, in the configured language, tersely (≈8 words per half). If a side is genuinely absent write '—' rather than dropping it. Do NOT invent separate pros/cons fields: Option accepts no other keys.",
163
+ }),
164
+ ),
160
165
  recommended: Type.Optional(Type.Boolean({ description: "Mark exactly one recommended option; put it first. Never embed (推荐)/(recommended) text in labels — the UI renders the ★ marker automatically" })),
161
166
  },
162
167
  { additionalProperties: false },
@@ -165,7 +170,10 @@ export const Option = Type.Object(
165
170
  export const BatchQuestionParams = Type.Object(
166
171
  {
167
172
  question: Type.String({ description: "The question to ask, in the configured language" }),
168
- options: Type.Array(Option, { description: "Ordered options: recommended first, alternatives next. Do not include Other or Auto-complete yourself." }),
173
+ options: Type.Array(Option, {
174
+ description:
175
+ "Ordered options: recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself.",
176
+ }),
169
177
  allowOther: Type.Optional(Type.Boolean({ description: "Offer free-form input for this question (default true)" })),
170
178
  autoComplete: Type.Optional(
171
179
  Type.Boolean({
@@ -182,7 +190,7 @@ export const BatchQuestionParams = Type.Object(
182
190
  export const AskChoiceParams = Type.Object(
183
191
  {
184
192
  question: Type.Optional(Type.String({ description: "The single question to ask, in the configured language (mutually exclusive with questions)." })),
185
- options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Do not include Other or Auto-complete yourself." })),
193
+ options: Type.Optional(Type.Array(Option, { description: "Ordered options (single-question form): recommended first, alternatives next. Every option's description states its advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the configured language. Do not include Other or Auto-complete yourself." })),
186
194
  questions: Type.Optional(
187
195
  Type.Array(BatchQuestionParams, {
188
196
  description:
@@ -586,16 +594,24 @@ export function registerAskChoiceTool(pi: ExtensionAPI): void {
586
594
  name: "ask_choice",
587
595
  label: "Ask Choice",
588
596
  description:
589
- "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt.",
597
+ "Ask the user planning or refinement questions as numbered choice prompts: recommended option first, alternatives next, then Other and Auto-complete. Two shapes: questions: [...] (2-8 questions) opens ONE tabbed multiple-choice form with a submit page — use it to batch a round of questions (≤8), then think about the answers and follow up in later calls (phased questioning stays agent-driven); question + options asks one question at a time (classic flow). Use ask_choice for every user-facing planning question, the final scope confirmation, refinement-mode questions, language/role/model settings, and the execution handoff. Scope confirmation and the execution handoff MUST stay single-question calls (autoComplete: false); batches reject autoComplete: false items and the termination/questionIds reserved for handoff. The optional trailing parameter swaps the trailing Auto-complete option to Auto-refine loop for the post-execution amelioration prompt. EVERY option you author — including the accept/execute handoff and the implementation-review setup questions — must set description to '✓ <advantage> / ✗ <drawback>' in the configured language, so the user can see what each option gains and what it costs. Other and Auto-complete are appended by this tool and need no description.",
590
598
  promptSnippet: "Ask structured planning questions with recommended/Other/Auto-complete ordering; batch ≤8 questions per form",
591
599
  promptGuidelines: [
592
600
  "Use ask_choice for every pi-plans question to the user instead of plain-text questions; it enforces option ordering and records decisions.",
593
601
  "Batch a round's questions into one ask_choice call (questions: [...], 2-8 items) instead of asking one at a time, then think after the answers and follow up with later calls. Scope confirmation and execution handoff are always separate single-question calls (autoComplete: false).",
602
+ "Give every option you author a description of the form '✓ <advantage> / ✗ <drawback>' — the user's whole point is seeing what each option wins and what it costs, in the configured language. Keep each half terse (~8 words). Put both halves in the description string; there are no separate pros/cons fields, and Other/Auto-complete are added by the tool.",
594
603
  ],
595
604
  parameters: AskChoiceParams,
596
605
  executionMode: "sequential",
597
606
 
598
607
  async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
608
+ // R-13 (defense-in-depth): a delegated executor child has no user to
609
+ // answer — refuse instead of blocking a headless run on a UI prompt.
610
+ if (process.env.PI_PLANS_EXECUTOR === "1") {
611
+ throw new Error(
612
+ "ask_choice is unavailable in a delegated executor session: no interactive user. Decide autonomously, proceed, and record the deviation in your final summary.",
613
+ );
614
+ }
599
615
  // 0.4.0 batch mode: one tabbed form for a whole round of questions
600
616
  // (2-8). The single-question path below is untouched (C-004).
601
617
  // F-005 (impl review r1): ambiguous shapes fail loudly instead of
@@ -12,12 +12,16 @@ import {
12
12
  getExecution,
13
13
  resumeActiveExecution,
14
14
  startExecution,
15
+ type ExecutionRuntime,
15
16
  } from "../src/exec.ts";
16
17
  import { disableAutoComplete } from "../src/autocomplete.ts";
17
18
  import { isAutoApproveEnabled } from "../src/auto-approve.ts";
18
19
  import { latestPlanVersion, parseChecklist, parseImplItems } from "../src/plan.ts";
19
- import { normalizeWorkdir, readActive } from "../src/state.ts";
20
- import { resolveActiveRun } from "../src/run-context.ts";
20
+ import { normalizeWorkdir, recordDecision, type RunSummary } from "../src/state.ts";
21
+ import { bindRun, resolveActiveRun } from "../src/run-context.ts";
22
+ import { executionCandidates, resolveCommandRun } from "../src/run-picker.ts";
23
+ import { collectModelSelectors, modelSelectorOf } from "../src/config-command.ts";
24
+ import { resolveUiLanguage } from "../src/ui-language.ts";
21
25
 
22
26
 
23
27
  const ExecutePlanParams = Type.Object({
@@ -39,23 +43,30 @@ export async function executeHandoff(
39
43
  ctx: ExtensionContext,
40
44
  planPathArg?: string,
41
45
  workdirArg?: string,
46
+ signal?: AbortSignal,
42
47
  ): Promise<HandoffOutcome> {
43
48
  const workdir = normalizeWorkdir(workdirArg ?? ctx.cwd);
44
49
 
45
50
  let planPath: string | null = null;
51
+ let chosenRun: RunSummary | null = null;
46
52
  if (planPathArg) {
47
53
  planPath = path.resolve(workdir, planPathArg.replace(/^@/, ""));
48
54
  } else {
49
- const active = resolveActiveRun(ctx.sessionManager, workdir);
50
- if (!active) {
55
+ // v0.6.0 (R-3): pick the run explicitly when several execution
56
+ // candidates coexist; binding-first, single candidate stays direct.
57
+ chosenRun = await resolveCommandRun(
58
+ { cwd: workdir, sessionManager: ctx.sessionManager, ui: ctx.ui },
59
+ { candidates: executionCandidates(workdir), title: "Execute which run?" },
60
+ );
61
+ if (!chosenRun) {
51
62
  return {
52
63
  status: "error",
53
- message: "No plan path given and no active planning run found. Pass planPath or start a run first.",
64
+ message: "No plan path given and no executable planning run found. Pass planPath or start a run first.",
54
65
  };
55
66
  }
56
- const latest = latestPlanVersion(active.artifact_dir);
67
+ const latest = latestPlanVersion(chosenRun.artifact_dir);
57
68
  if (!latest) {
58
- return { status: "error", message: `No PLAN_vN.md found in ${active.artifact_dir}` };
69
+ return { status: "error", message: `No PLAN_vN.md found in ${chosenRun.artifact_dir}` };
59
70
  }
60
71
  planPath = latest.path;
61
72
  }
@@ -99,17 +110,87 @@ export async function executeHandoff(
99
110
  return { status: "declined", message: "User declined execution. Stay in planning; ask how to proceed.", planPath };
100
111
  }
101
112
 
102
- await startExecution(getCurrentApi(), ctx, planPath, items, implItems);
113
+ // Attribute the handoff to the picked run before any state transition so
114
+ // the approval checkpoint and status flip land on the run the user chose.
115
+ if (chosenRun) bindRun(ctx.sessionManager, workdir, chosenRun.run_id);
116
+
117
+ // v0.6.0 (R-8): runtime question — current session (recommended) or a
118
+ // delegated executor on another model. Skipped under auto-approve/no-UI.
119
+ const runtime = await chooseExecutionRuntime(ctx, workdir, autoApprove, chosenRun);
120
+
121
+ await startExecution(getCurrentApi(), ctx, planPath, items, implItems, { runtime, signal });
103
122
  const scopeNote = implItems.length ? ` Tracking ${implItems.length} implementation item(s).` : "";
104
123
  const autoNote = autoApprove ? "[auto-approve] " : "";
124
+ const runtimeNote = runtime === "current-session" ? "" : ` Delegated to executor model ${runtime.modelSelector}; progress mirrors in the overlay.`;
105
125
  return {
106
126
  status: "executing",
107
127
  planPath,
108
128
  itemCount: items.length,
109
- message: `${autoNote}Execution approved. ${items.length} verifier item(s) queued; implement in dependency order and mark verified items with [DONE:VC-xxx].${scopeNote}`,
129
+ message: `${autoNote}Execution approved. ${items.length} verifier item(s) queued; implement in dependency order and mark verified items with [DONE:VC-xxx].${scopeNote}${runtimeNote}`,
110
130
  };
111
131
  }
112
132
 
133
+ const MODEL_SELECTOR_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*\/[A-Za-z0-9][A-Za-z0-9._-]*$/;
134
+
135
+ /**
136
+ * R-8: ask where the execution runs. Uses ctx.ui.select directly (ask_choice
137
+ * is a tool and cannot be invoked from tool/command context). Under
138
+ * auto-approve or no-UI the question is skipped: current session, decision
139
+ * recorded with the [auto-approve] annotation convention.
140
+ */
141
+ async function chooseExecutionRuntime(
142
+ ctx: ExtensionContext,
143
+ workdir: string,
144
+ autoApprove: boolean,
145
+ chosenRun: RunSummary | null,
146
+ ): Promise<ExecutionRuntime> {
147
+ const record = (answer: string, source: "user" | "auto-complete", question: string, options: string[]): void => {
148
+ const runId = chosenRun?.run_id ?? resolveActiveRun(ctx.sessionManager, workdir)?.run_id ?? null;
149
+ if (!runId) return;
150
+ try {
151
+ recordDecision(workdir, runId, {
152
+ question,
153
+ options,
154
+ answer,
155
+ answer_source: source,
156
+ });
157
+ } catch {
158
+ /* decision audit trail is best-effort */
159
+ }
160
+ };
161
+ if (autoApprove || !ctx.hasUI) {
162
+ record("current session [auto-approve]", "auto-complete", "Execution runtime", ["current session", "switch model"]);
163
+ return "current-session";
164
+ }
165
+ const lang = resolveUiLanguage(workdir);
166
+ const currentLabel = lang === "zh" ? "使用当前会话(推荐)" : "Use the current session (recommended)";
167
+ const switchLabel = lang === "zh" ? "切换至其他模型…" : "Switch to another model…";
168
+ const title = lang === "zh" ? "执行运行时" : "Execution runtime";
169
+ const first = await ctx.ui.select(title, [currentLabel, switchLabel]);
170
+ if (first === undefined || first === currentLabel) {
171
+ record("current session", "user", "Execution runtime", [currentLabel, switchLabel]);
172
+ return "current-session";
173
+ }
174
+ // Model picker: switch targets exclude the current selector by design
175
+ // (option 1 IS the current session).
176
+ const currentSelector = modelSelectorOf(ctx.model);
177
+ const targets = collectModelSelectors(ctx, currentSelector);
178
+ const otherLabel = lang === "zh" ? "其他(输入 provider/model)…" : "Other (type provider/model)…";
179
+ const modelTitle = lang === "zh" ? "切换至哪个模型执行?" : "Switch to which model?";
180
+ let modelPick = await ctx.ui.select(modelTitle, [...targets, otherLabel]);
181
+ if (modelPick === otherLabel) {
182
+ const typed = await ctx.ui.input(modelTitle, "provider/model");
183
+ modelPick = typed && MODEL_SELECTOR_RE.test(typed.trim()) ? typed.trim() : undefined;
184
+ }
185
+ if (modelPick === undefined || !MODEL_SELECTOR_RE.test(modelPick)) {
186
+ // Cancelled or invalid: fall back to the current session, recorded.
187
+ record("current session (model switch cancelled)", "user", "Execution runtime", [currentLabel, switchLabel]);
188
+ return "current-session";
189
+ }
190
+ record(`switch model: ${modelPick}`, "user", "Execution runtime", [currentLabel, switchLabel, ...targets, otherLabel]);
191
+ return { modelSelector: modelPick };
192
+ }
193
+
113
194
  /** The user command may resume an approved execution; the tool always asks. */
114
195
  export async function executeCommand(ctx: ExtensionContext, planPathArg?: string): Promise<HandoffOutcome> {
115
196
  const activeExecution = getExecution();
@@ -143,12 +224,12 @@ export function registerExecutePlanTool(pi: ExtensionAPI): void {
143
224
  name: "execute_plan",
144
225
  label: "Execute Plan",
145
226
  description:
146
- "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters plan-execution mode: the extension injects the remaining Verifier Checklist every turn, tracks [DONE:VC-xxx] markers, and completes when every item passes. Only call after the user chose 'Execute this plan now' at the handoff question.",
227
+ "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then asks which runtime executes the plan — the current session (recommended) or a delegated executor subagent on another model (>=3 switch targets listed; the child writes natively and reports [DONE:VC-xxx] markers the parent tracks). Either way the extension tracks Verifier-Checklist progress. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
147
228
  promptSnippet: "Hand an accepted plan off to the tracked execution loop",
148
229
  parameters: ExecutePlanParams,
149
230
 
150
- async execute(_toolCallId, params, _signal, _onUpdate, ctx) {
151
- const outcome = await executeHandoff(ctx, params.planPath, params.workdir);
231
+ async execute(_toolCallId, params, signal, _onUpdate, ctx) {
232
+ const outcome = await executeHandoff(ctx, params.planPath, params.workdir, signal);
152
233
  if (outcome.status === "error") throw new Error(outcome.message);
153
234
  return {
154
235
  content: [{ type: "text", text: outcome.message }],
@@ -277,6 +277,10 @@ function createGraphReadTool(cwd: string) {
277
277
  };
278
278
  };
279
279
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
280
+ // Delegated executor children (PI_PLANS_EXECUTOR=1) always use the
281
+ // native tools: DB-first staging would never be materialized inside
282
+ // the child (no code_graph in its allowlist), so writes must hit disk.
283
+ if (process.env.PI_PLANS_EXECUTOR === "1") return native(null);
280
284
  if (mode === "off") return native(null);
281
285
  if (mode === "config-unavailable") return native("config read failed");
282
286
  const ensured = await ensureRuntime(ctx.cwd, ctx);
@@ -326,6 +330,8 @@ function createGraphWriteTool(cwd: string) {
326
330
  };
327
331
  };
328
332
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
333
+ // Delegated executor children bypass DB-first staging (see read tool).
334
+ if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
329
335
  if (mode === "off") return stage(null);
330
336
  if (mode === "config-unavailable") return stage("config read failed");
331
337
  const ensured = await ensureRuntime(ctx.cwd, ctx);
@@ -375,6 +381,8 @@ function createGraphEditTool(cwd: string) {
375
381
  };
376
382
  };
377
383
  const mode: GraphMode = resolveGraphMode(ctx.cwd);
384
+ // Delegated executor children bypass DB-first staging (see read tool).
385
+ if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);
378
386
  if (mode === "off") return stage(null);
379
387
  if (mode === "config-unavailable") return stage("config read failed");
380
388
  const ensured = await ensureRuntime(ctx.cwd, ctx);
package/tools/plans.ts CHANGED
@@ -296,7 +296,7 @@ export function registerPlansTool(pi: ExtensionAPI): void {
296
296
  name: "plans",
297
297
  label: "Plans",
298
298
  description:
299
- "Manage pi-plans planning state in the target workspace: init/show config, set language and planning docs root plus reviewer/criticizer roles and the code-graph enabled flag, start planning runs, record decisions/refs/subagents, and update run status. State lives in .git/pi_plans/ inside the resolved git common dir. Actions: init, show, set-language, set-artifact-root, set-refs-root, set-graph-enabled, set-role, start-run, set-status, final-commit, record-decision, record-ref, record-subagent.",
299
+ "Manage pi-plans planning state in the target workspace: init/show config, set language and planning docs root plus reviewer/criticizer roles and the code-graph enabled flag, start planning runs, record decisions/refs/subagents, and update run status. Multiple concurrent runs per workdir are supported (registry-derived from runs/; sessions bind to their run). State lives in .git/pi_plans/ inside the resolved git common dir. Actions: init, show, set-language, set-artifact-root, set-refs-root, set-graph-enabled, set-role, start-run, set-status, final-commit, record-decision, record-ref, record-subagent.",
300
300
  promptSnippet: "Manage pi-plans planning state, runs, and ledgers",
301
301
  parameters: PlansParams,
302
302