pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
@@ -17,11 +17,11 @@ export const REVIEWER_LENSES: readonly ReviewerLane[] = [
17
17
  { id: "verification", lens: "verification rigor, risks, and evidence gaps" },
18
18
  ] as const;
19
19
 
20
- function buildSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
21
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
20
+ function buildSharedHeader(opts: RefinePromptInput): string {
21
+ const lensLine = opts.lens ? `\nReview lens: ${opts.lens}.` : "";
22
22
  const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
23
23
  const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
24
- return `Goal: ${role === "reviewer" ? "review the plan against the repository" : "stress-test the plan's assumptions"}.
24
+ return `Goal: review the plan against the repository and surface what needs the user's judgment.
25
25
 
26
26
  Target: ${opts.planPath}
27
27
 
@@ -30,23 +30,6 @@ Authority boundary: read-only analysis only. Do not edit, write, delete, commit,
30
30
  Evidence: inspect the repository with read, grep, find, and ls before judging the plan.${lensLine}${focusLine}${contextLine}`;
31
31
  }
32
32
 
33
- /** Shared header for the post-execution implementation review: the
34
- * accepted plan is the contract, the IMPLEMENTATION in the worktree is
35
- * under review. Findings must anchor to the plan's goals/acceptance
36
- * criteria and explicitly assess delivery maturity. */
37
- function buildImplementationSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
38
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
39
- const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
40
- const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
41
- return `Goal: ${role === "reviewer" ? "review the implemented result in the worktree against the plan" : "stress-test the implemented result's assumptions"}.
42
-
43
- Target: ${opts.planPath} (the accepted plan; the IMPLEMENTATION in the worktree is under review)
44
-
45
- Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
46
-
47
- Evidence: inspect the repository with read, grep, find, and ls before judging the implementation. Judge the implementation against the plan's goals, verifier checklist, and acceptance criteria. Explicitly assess delivery maturity: did the executor ship a minimal MVP only, or refine for long-term growth (no stopgaps, long-term architectural decisions, missing tests, technical debt, production readiness)? Calibrate severity accordingly. Out-of-scope improvement ideas are low severity by default and must not be forced into findings.${lensLine}${focusLine}${contextLine}`;
48
- }
49
-
50
33
  export function reviewerLanes(count: number): ReviewerLane[] {
51
34
  if (count === 3) return [...REVIEWER_LENSES];
52
35
  if (count === 2) return [...REVIEWER_LENSES.slice(0, 2)];
@@ -54,47 +37,22 @@ export function reviewerLanes(count: number): ReviewerLane[] {
54
37
  }
55
38
 
56
39
  export function buildReviewerTask(opts: RefinePromptInput): string {
57
- return `${buildSharedHeader("reviewer", opts)}
58
-
59
- Success criteria: return evidence-backed findings or explicitly say the plan holds up.
60
-
61
- Output: Markdown, highest severity first. For each finding use this shape:
62
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
63
-
64
- Surface at most five high-priority findings; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
65
-
66
- Plan file: ${opts.planPath}
67
-
68
- ---8<--- PLAN CONTENT ---8<---
69
- ${opts.planText}
70
- ---8<--- END PLAN CONTENT ---8<---`;
71
- }
40
+ return `${buildSharedHeader(opts)}
72
41
 
73
- export function buildCriticizerTask(opts: RefinePromptInput): string {
74
- return `${buildSharedHeader("criticizer", opts)}
42
+ Success criteria: return evidence-backed findings, plus the questions only the user can settle. If the plan holds up, say so explicitly and list what you checked.
75
43
 
76
- Success criteria: return concrete, answerable questions only; never rewrite the plan.
44
+ Output: Markdown with exactly two top-level parts, in this order.
77
45
 
78
- Output: Markdown in exactly this shape:
79
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
80
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
46
+ ## Findings
81
47
 
82
- Plan file: ${opts.planPath}
83
-
84
- ---8<--- PLAN CONTENT ---8<---
85
- ${opts.planText}
86
- ---8<--- END PLAN CONTENT ---8<---`;
87
- }
88
-
89
- export function buildImplementationReviewerTask(opts: RefinePromptInput): string {
90
- return `${buildImplementationSharedHeader("reviewer", opts)}
48
+ Highest severity first. For each finding use this shape:
49
+ - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
91
50
 
92
- Success criteria: return evidence-backed findings or explicitly say the implementation holds up.
51
+ Surface at most five high-priority findings; list lower-severity findings after them. Write "None." when there are none.
93
52
 
94
- Output: Markdown, highest severity first. For each finding use this shape:
95
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003) or files; evidence: repo path/command that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
53
+ ## Questions
96
54
 
97
- Surface at most five high-priority findings; list lower-severity findings after them. If the implementation holds up, say so explicitly and list what you checked.
55
+ At most five numbered questions (\`Q-1\`, \`Q-2\`, …) covering everything that needs the user's decision before the plan can be safely revised — hidden trade-offs, undetermined semantics, accept/reject calls on findings marked needs-discussion. Each question gets one line of why it matters, phrased so a user with repo access can answer it concretely. Never rhetorical; never questions the repository itself already answers. Stop earlier if nothing genuinely needs the user.
98
56
 
99
57
  Plan file: ${opts.planPath}
100
58
 
@@ -144,7 +102,7 @@ Local path (your working directory): ${opts.localPath}
144
102
 
145
103
  Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents. Stay inside the reference directory.
146
104
 
147
- Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite files as <relative-path>:<line> for every claim; quote only what you verified.${contextLine}${languageLine}
105
+ Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite evidence for every claim in the medium's format — code: <relative-path>:<line>; papers: section/theorem/table numbers with a short quote; blogs/docs: the heading or quoted passage. Quote only what you verified. Medium-aware deep-read: repos go through entry points, core modules, tests, and configuration; papers through claims, method, limitations, and experiments; blogs/docs through technique, measurements, and caveats. Theoretical grounding counts — an algorithm, a formal property, or a measured tradeoff is as adoptable as an implementation pattern.${contextLine}${languageLine}
148
106
 
149
107
  Success criteria: a structured analysis the main agent can paste into REF_ANALYSIS.md and turn into adoption questions.
150
108
 
@@ -157,23 +115,7 @@ Section contracts:
157
115
  - Key Mechanisms And Design Tradeoffs: the mechanisms that make it work and the tradeoffs they embody.
158
116
  - Adoptable Ideas For The Target Repo: concrete, portable ideas ranked by expected value; name the target-repo surface each would touch.
159
117
  - Pitfalls And Anti-Patterns: what to avoid when borrowing; failure modes the reference itself documents or exhibits.
160
- - Evidence Citations: the file:line references backing the claims above.
118
+ - Evidence Citations: the evidence references backing the claims above (file:line for code; section/theorem/table + quote for papers; heading/quote for blogs and docs).
161
119
  - Coverage: which parts of the reference you actually read versus skipped.
162
120
  - Evidence Gaps: what you could not determine from the reference alone.`;
163
121
  }
164
-
165
- export function buildImplementationCriticizerTask(opts: RefinePromptInput): string {
166
- return `${buildImplementationSharedHeader("criticizer", opts)}
167
-
168
- Success criteria: return concrete, answerable questions only; never rewrite the plan or the implementation.
169
-
170
- Output: Markdown in exactly this shape:
171
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
172
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the implementation genuinely holds.
173
-
174
- Plan file: ${opts.planPath}
175
-
176
- ---8<--- PLAN CONTENT ---8<---
177
- ${opts.planText}
178
- ---8<--- END PLAN CONTENT ---8<---`;
179
- }
@@ -10,11 +10,30 @@
10
10
  const ELLIPSIS = "…";
11
11
 
12
12
  /**
13
- * ECMA-48 CSI sequence: ESC [ parameter bytes (0x30-0x3F), intermediate bytes
14
- * (0x20-0x2F), final byte (0x40-0x7E). Matched atomically so styling payloads
15
- * never leak into width math and are never split mid-sequence.
13
+ * ECMA-48 escape sequences. ALL of them render at zero width, so styling and
14
+ * control payloads must never leak into width math nor be split mid-sequence.
15
+ *
16
+ * Three families are recognised:
17
+ * - CSI: `ESC [ params intermediates final` (SGR colours, cursor moves)
18
+ * - String-terminated: `ESC ] _ P X ^` … `BEL`|`ST` (OSC, APC, DCS, SOS, PM)
19
+ * - Simple: `ESC` intermediates `final` (`ESC c`, `ESC (B`, `ESC 7`)
20
+ *
21
+ * The string family matters beyond colour: pi's `CURSOR_MARKER` is an APC
22
+ * sequence (`ESC _ pi:c BEL`). Counting its payload as visible text made every
23
+ * row carrying the marker — e.g. a focused search Input — measure several
24
+ * columns too wide, which pushed that row's right-hand border out of
25
+ * alignment. The terminator is required: a well-formed sequence always has
26
+ * one, and refusing malformed ones keeps the fallback (the simple family)
27
+ * from swallowing real text.
16
28
  */
17
- const CSI_PATTERN = /\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]/g;
29
+ const ESCAPE_PATTERN = new RegExp(
30
+ [
31
+ "\\x1b\\[[\\x30-\\x3f]*[\\x20-\\x2f]*[\\x40-\\x7e]", // CSI
32
+ "\\x1b[\\]PX^_][^\\x07\\x1b]*(?:\\x07|\\x1b\\\\)", // OSC / APC / DCS / SOS / PM
33
+ "\\x1b[\\x20-\\x2f]*[\\x30-\\x7e]", // simple escapes
34
+ ].join("|"),
35
+ "g",
36
+ );
18
37
 
19
38
  interface AnsiPart {
20
39
  kind: "csi" | "text";
@@ -24,7 +43,7 @@ interface AnsiPart {
24
43
  function splitAnsi(text: string): AnsiPart[] {
25
44
  const parts: AnsiPart[] = [];
26
45
  let last = 0;
27
- for (const match of text.matchAll(CSI_PATTERN)) {
46
+ for (const match of text.matchAll(ESCAPE_PATTERN)) {
28
47
  const start = match.index ?? 0;
29
48
  if (start > last) parts.push({ kind: "text", value: text.slice(last, start) });
30
49
  parts.push({ kind: "csi", value: match[0] });
@@ -1,6 +1,6 @@
1
1
  import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
2
2
 
3
- export type RefineOverlayRole = "reviewer" | "criticizer" | "refs";
3
+ export type RefineOverlayRole = "reviewer" | "refs";
4
4
  export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
5
5
  export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
6
6
 
package/src/refine-ui.ts CHANGED
@@ -166,7 +166,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
166
166
  const complete = lanes.filter((lane) => lane.status === "complete").length;
167
167
  const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
168
168
  const running = lanes.filter((lane) => lane.status === "running").length;
169
- const title = role === "reviewer" ? "Reviewer" : role === "refs" ? "Refs" : "Criticizer";
169
+ const title = role === "reviewer" ? "Reviewer" : "Refs";
170
170
  const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
171
171
  const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
172
172
  return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
@@ -12,14 +12,14 @@
12
12
  * - R-007: after every dialog the world is re-checked; one kickoff max.
13
13
  */
14
14
 
15
- import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
15
+ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
16
16
  import * as fs from "node:fs";
17
17
  import { existsSync } from "node:fs";
18
18
  import * as path from "node:path";
19
19
  import { loadExecutionFromCheckpoint } from "./exec.ts";
20
- import { bindRun } from "./run-context.ts";
20
+ import { bindRun, boundRunId } from "./run-context.ts";
21
21
  import { acquireOwnership, OwnershipError, releaseOwnership } from "./run-ownership.ts";
22
- import { loadConfig, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
22
+ import { loadConfig, resolveArtifactRoot, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
23
23
  import {
24
24
  applyMigration,
25
25
  mutateCheckpoint,
@@ -35,12 +35,12 @@ import {
35
35
  reconcileCheckpointWithLedger,
36
36
  type ResumeCandidate,
37
37
  } from "./resume.ts";
38
+ import { messaging } from "./messaging.ts";
38
39
 
39
40
  /** One kickoff per invocation; repeat invocations are blocked by the idle check. */
40
41
  let inFlight = false;
41
42
 
42
43
  export async function resumePlansCommand(
43
- pi: ExtensionAPI,
44
44
  ctx: ExtensionContext,
45
45
  baseDir: string,
46
46
  ): Promise<void> {
@@ -50,13 +50,13 @@ export async function resumePlansCommand(
50
50
  }
51
51
  inFlight = true;
52
52
  try {
53
- await run(pi, ctx, baseDir);
53
+ await run(ctx, baseDir);
54
54
  } finally {
55
55
  inFlight = false;
56
56
  }
57
57
  }
58
58
 
59
- async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Promise<void> {
59
+ async function run(ctx: ExtensionContext, baseDir: string): Promise<void> {
60
60
  if (!ctx.hasUI) {
61
61
  ctx.ui.notify?.("/resume-plans needs an interactive session (TUI/RPC); print/json cannot resume.", "warning");
62
62
  return;
@@ -73,7 +73,11 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
73
73
  return;
74
74
  }
75
75
 
76
- let candidate = pickDefaultCandidate(ctx.cwd, candidates);
76
+ // v0.6.0 (D-1): a session-bound resumable run resumes directly; the
77
+ // active-pointer auto-win is gone — ambiguity opens the descriptive form.
78
+ const bound = boundRunId(ctx.sessionManager, ctx.cwd);
79
+ const boundCandidate = bound ? candidates.find((candidate) => candidate.runId === bound) ?? null : null;
80
+ let candidate = boundCandidate ?? pickDefaultCandidate(ctx.cwd, candidates);
77
81
  if (candidate === null) {
78
82
  const labels = candidates.map((entry, index) => {
79
83
  const cross = entry.crossWorktree ? " · cross-worktree" : "";
@@ -101,7 +105,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
101
105
  }
102
106
  if (candidate.checkpointStatus === "corrupt") {
103
107
  ctx.ui.notify(
104
- `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi_plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
108
+ `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi-plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
105
109
  "error",
106
110
  );
107
111
  return;
@@ -155,7 +159,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
155
159
  }
156
160
  }
157
161
 
158
- const brief = await buildBrief(pi, ctx, baseDir, candidate);
162
+ const brief = await buildBrief(ctx, baseDir, candidate);
159
163
  if (brief === null) {
160
164
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
161
165
  return; // buildBrief reported the specific problem
@@ -163,7 +167,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
163
167
 
164
168
  // Exactly one kickoff (R-007). The idle re-check happens above and in
165
169
  // buildBrief's dialogs; the message itself starts the continuation.
166
- await pi.sendUserMessage(brief.text);
170
+ await messaging().sendUserMessage(brief.text);
167
171
  ctx.ui.notify(`Resumed ${candidate.runId} (${brief.phaseLabel}).`, "info");
168
172
  } catch (error) {
169
173
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
@@ -185,13 +189,12 @@ export function migrateRunIntoCurrentWorktree(workdir: string, candidate: Resume
185
189
  const stateRoot = loadConfigShared(workdir);
186
190
  if (stateRoot === null) return null;
187
191
  const config = loadConfig(stateRoot);
188
- let artifactRoot = config.artifact_root;
189
- if (!path.isAbsolute(artifactRoot)) artifactRoot = path.resolve(workdir, artifactRoot);
192
+ const artifactRoot = resolveArtifactRoot(workdir, config.artifact_root);
190
193
  const sourceDir = candidate.run.artifact_dir;
191
194
  if (!existsSync(sourceDir)) return null;
192
195
  // Artifacts already in a shared location stay put.
193
196
  const gitRoot = findGitCommonDir(workdir);
194
- if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi_plans"))) return sourceDir;
197
+ if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi-plans"))) return sourceDir;
195
198
  const targetDir = path.join(artifactRoot, path.basename(sourceDir));
196
199
  for (const entry of fs.readdirSync(sourceDir, { withFileTypes: true })) {
197
200
  if (!entry.isFile()) continue;
@@ -257,46 +260,7 @@ export interface ResumeBrief {
257
260
  text: string;
258
261
  }
259
262
 
260
- /** D-5 crash-window recovery: rebuild the implementation-review loop
261
- * configuration from the checkpoint plus the decisions ledger. Answers land
262
- * in the ledger first (stable questionIds), the combined record-checkpoint
263
- * second; between the two, this resolver reconstructs what is known and
264
- * reports which questions are still missing. Exported for tests. */
265
- export function resolveImplReviewConfig(
266
- review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } | undefined,
267
- ledger: { questionId?: string; answer?: string }[],
268
- ): {
269
- condition: string | undefined;
270
- conditionFromLedger: boolean;
271
- reviewerCount: number | undefined;
272
- reviewerCountFromLedger: boolean;
273
- missing: Array<"termination-condition" | "impl-review-reviewer-count">;
274
- } {
275
- const latest = (id: string): string | undefined =>
276
- [...ledger].reverse().find((entry) => entry.questionId === id)?.answer;
277
- const ledgerCondition = latest("termination-condition");
278
- const ledgerCountRaw = latest("impl-review-reviewer-count");
279
- const ledgerCount = ledgerCountRaw === undefined ? undefined : Number.parseInt(ledgerCountRaw, 10);
280
- const ledgerCountValid = ledgerCount !== undefined && Number.isInteger(ledgerCount) && ledgerCount >= 1 && ledgerCount <= 3;
281
- const condition = review?.terminationCondition ?? ledgerCondition;
282
- const reviewerCount = review?.reviewerCount ?? (ledgerCountValid ? ledgerCount : undefined);
283
- // Both questions are per-run configuration: a count is missing whenever it
284
- // is unknown (including legacy checkpoints persisted before 0.5.4),
285
- // independent of where the condition came from.
286
- const missing: Array<"termination-condition" | "impl-review-reviewer-count"> = [];
287
- if (condition === undefined) missing.push("termination-condition");
288
- if (reviewerCount === undefined) missing.push("impl-review-reviewer-count");
289
- return {
290
- condition,
291
- conditionFromLedger: review?.terminationCondition === undefined && ledgerCondition !== undefined,
292
- reviewerCount,
293
- reviewerCountFromLedger: review?.reviewerCount === undefined && ledgerCountValid,
294
- missing,
295
- };
296
- }
297
-
298
263
  async function buildBrief(
299
- pi: ExtensionAPI,
300
264
  ctx: ExtensionContext,
301
265
  baseDir: string,
302
266
  candidate: ResumeCandidate,
@@ -323,7 +287,7 @@ async function buildBrief(
323
287
  bindRun(ctx.sessionManager, ctx.cwd, runId);
324
288
 
325
289
  if (cp.phase === "executing") {
326
- const load = loadExecutionFromCheckpoint(pi, ctx, runId);
290
+ const load = loadExecutionFromCheckpoint(ctx, runId);
327
291
  if (load.status === "loaded") {
328
292
  if (run.status === "stopped" || run.status === "accepted") {
329
293
  try {
@@ -334,14 +298,22 @@ async function buildBrief(
334
298
  }
335
299
  const doneList = (load.doneVcIds ?? []).join(", ") || "none";
336
300
  const reverify = load.reverifyAll
337
- ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously verified VC must be re-verified before new work counts. Historically verified (evidence only): ${doneList}.`
301
+ ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
338
302
  : `\nPreviously verified and still valid: ${doneList}.`;
339
- const paused = load.pausedReason ? `\nExecution was paused: ${load.pausedReason}. Continue from where it stopped.` : "";
303
+ const paused = load.pausedReason ? `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.` : "";
304
+ const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
340
305
  return {
341
306
  phaseLabel: "executing",
342
- text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}\nFollow the execution-loop contract: implement in dependency order, verify each VC, and mark completions with [DONE:VC-xxx]. The remaining checklist is injected each turn.`,
307
+ text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\nFollow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the completion auditor verify the checks. The current wave and remaining tasks are injected each turn.`,
343
308
  };
344
309
  }
310
+ if (load.legacyDelegate) {
311
+ ctx.ui.notify(
312
+ `Cannot resume ${runId} directly: it was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Run /plans-execute to re-approve the handoff; execution restarts from the first task (0.6.0 progress cannot map onto the task tree).`,
313
+ "warning",
314
+ );
315
+ return null;
316
+ }
345
317
  if (load.status === "plan-missing" || load.status === "plan-mismatch") {
346
318
  ctx.ui.notify(`Cannot resume ${runId}: ${load.error ?? "plan file missing"}.`, "error");
347
319
  return null;
@@ -355,69 +327,19 @@ async function buildBrief(
355
327
  }
356
328
 
357
329
  if (cp.phase === "implementation-review") {
358
- const review = cp.implementationReview;
359
- const condition = review?.terminationCondition;
360
- const lines: string[] = [
361
- `[PI-PLANS RESUME] Implementation review of run ${runId} continues in this session.`,
362
- `Plan: ${cp.plan?.path ?? "(unknown)"}`,
363
- ];
364
- // D-5 crash-window recovery: both config answers live in the decisions
365
- // ledger under stable questionIds. Rebuild from the ledger, ask ONLY the
366
- // missing question(s), then persist BOTH in one combined write. The
367
- // re-ask guard on the checkpoint (termination condition already
368
- // configured) rejects duplicate writes, so the combined write happens
369
- // only while the field is genuinely absent.
370
- const ledger = readDecisionLedger(ctx.cwd, runId);
371
- const resolved = resolveImplReviewConfig(review, ledger);
372
- const effectiveCondition = resolved.condition;
373
- const effectiveCount = resolved.reviewerCount;
374
- if (effectiveCondition === undefined) {
375
- lines.push(
376
- `The termination condition was never chosen. Ask it now via ask_choice (autoComplete: false, questionId: "termination-condition"): "How should the implementation-review loop terminate?" Options: goal wait (recommended) / until no high-severity finding (hard cap 5 rounds) / 1 / 2 / 3 rounds.`,
377
- );
378
- } else {
379
- lines.push(`Termination condition: ${effectiveCondition}${resolved.conditionFromLedger ? " (recovered from the decisions ledger)" : ""}`);
380
- }
381
- if (review?.reviewerCount === undefined) {
382
- if (effectiveCount !== undefined && Number.isInteger(effectiveCount) && effectiveCount >= 1 && effectiveCount <= 3) {
383
- lines.push(`Reviewer count: ${effectiveCount} (recovered from the decisions ledger; not yet persisted).`);
384
- } else {
385
- lines.push(
386
- `The reviewer count was never chosen. Ask it via ask_choice (autoComplete: false, allowOther: false, questionId: "impl-review-reviewer-count", digit labels "1"/"2"/"3"): "How many concurrent reviewers should each implementation-review round use?" — recommended default follows this run's skill: plan-big / plan-with-refs → 3, others → 1.`,
387
- );
388
- }
389
- } else {
390
- lines.push(`Reviewers per round: ${review.reviewerCount}`);
391
- }
392
- if (condition === undefined) {
393
- // Only when the checkpoint is still unconfigured: the combined write
394
- // (both answers) is one transaction. When only terminationCondition is
395
- // missing but a ledger answer exists for both, write both; when a
396
- // question is genuinely unanswered, ask first, then write.
397
- lines.push(
398
- `After both answers exist (asked now or recovered from the ledger), persist them in ONE call: plans record-checkpoint (checkpoint: { transition: "implementation-review-configured", terminationCondition: "<answer>", reviewerCount: <integer> }).`,
399
- );
400
- } else {
401
- lines.push(`Completed rounds in this worktree: ${review?.completedRounds ?? 0} (hard cap 5).`);
402
- }
403
- const currentRound = review?.currentRoundId
404
- ? cp.reviewRounds.find((round) => round.roundId === review.currentRoundId)
405
- : undefined;
406
- if (currentRound) {
407
- const done = currentRound.lanes.filter((lane) => lane.status === "complete").map((lane) => lane.laneId);
408
- const pending = currentRound.lanes.filter((lane) => lane.status !== "complete").map((lane) => lane.laneId);
409
- lines.push(
410
- `Round ${currentRound.roundId} is in flight — complete lanes: ${done.join(", ") || "none"}; pending/failed lanes: ${pending.join(", ") || "none"}. Resume it with refine (role: "reviewer", target: "implementation", resumeRoundId: "${currentRound.roundId}") so completed lanes are reused, never re-run.`,
411
- );
412
- } else {
413
- lines.push(
414
- `Start the next round with refine (role: "reviewer", target: "implementation", reviewers: ${review?.reviewerCount ?? effectiveCount ?? "<configured count>"}) — do NOT pass a resumeRoundId unless resuming an interrupted round; an omitted reviewers argument falls back to the run's configured reviewerCount.`,
415
- );
330
+ // v0.6.1 (D-018/D-020): the implementation-review loop is gone; a
331
+ // legacy 0.6.0 checkpoint in this phase maps to execution completed.
332
+ // The run flips to done so the status line and run registry agree.
333
+ try {
334
+ setRunStatus(ctx.cwd, runId, "done");
335
+ } catch {
336
+ /* best-effort */
416
337
  }
417
- lines.push(
418
- `Record boundaries with plans record-checkpoint: review-consolidated → implementation-round-finished per round; completed (with evidence) when the termination condition is met.`,
338
+ ctx.ui.notify(
339
+ `${runId} finished under the removed v0.6.0 implementation-review loop; mapped to done. Its historical acceptance stands; no review loop to resume.`,
340
+ "info",
419
341
  );
420
- return { phaseLabel: "implementation-review", text: lines.join("\n") };
342
+ return null;
421
343
  }
422
344
 
423
345
  // planning / reviewing: rebuild the workflow context (R-002/R-003).
@@ -508,19 +430,7 @@ async function buildLegacyBrief(
508
430
  `This run predates durable checkpoints: treat every VC as unverified and re-run the execution handoff (execute_plan or /plans-execute) for explicit approval before writing any code.`,
509
431
  );
510
432
  }
511
- if (candidate.run.status === "done") {
512
- const ok = await ctx.ui.confirm(
513
- "Finished run with review artifacts",
514
- `${candidate.runId} is marked done and has review records, but completion of the implementation-review loop cannot be proven for legacy runs. Resume the review loop anyway?`,
515
- );
516
- if (!ok) {
517
- ctx.ui.notify("Cancelled; nothing was changed.", "info");
518
- return null;
519
- }
520
- lines.push(
521
- `Resume the implementation-review loop: ask the termination condition (questionId: "termination-condition") if unknown, then run rounds with refine (target: "implementation").`,
522
- );
523
- }
524
433
  lines.push(`Continue only the missing work; never restart the interview from scratch.`);
525
434
  return { phaseLabel: candidate.phaseLabel, text: lines.join("\n") };
526
435
  }
436
+
package/src/resume.ts CHANGED
@@ -9,7 +9,7 @@
9
9
  import * as fs from "node:fs";
10
10
  import { existsSync, readFileSync } from "node:fs";
11
11
  import * as path from "node:path";
12
- import { getRun, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
12
+ import { getRun, listRuns, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
13
13
  import { loadCheckpoint, mutateCheckpoint, type WorkflowCheckpoint } from "./workflow-state.ts";
14
14
 
15
15
  export interface ResumeCandidate {
@@ -81,22 +81,17 @@ function planVersionOf(checkpoint: WorkflowCheckpoint | null, run: RunInfo): num
81
81
  /**
82
82
  * Enumerate resumable runs for the repo containing `workdir`. Corrupt
83
83
  * checkpoints are surfaced (not hidden) so the command can report them;
84
- * read errors never abort discovery of other runs.
84
+ * read errors never abort discovery of other runs. v0.6.0: enumeration is
85
+ * driven by the filesystem-derived registry (`listRuns`) so ordering and
86
+ * corrupt-run tolerance match every other multi-run surface.
85
87
  */
86
88
  export function listResumeCandidates(workdir: string): ResumeCandidate[] {
87
89
  const stateRoot = resolveStateRootOrNull(workdir);
88
90
  if (stateRoot === null) return [];
89
- const runsDir = path.join(stateRoot, "runs");
90
- if (!existsSync(runsDir)) return [];
91
91
  const active = readActive(workdir);
92
92
  const candidates: ResumeCandidate[] = [];
93
- let entries: string[] = [];
94
- try {
95
- entries = fs.readdirSync(runsDir);
96
- } catch {
97
- return [];
98
- }
99
- for (const runId of entries) {
93
+ for (const summary of listRuns(workdir)) {
94
+ const runId = summary.run_id;
100
95
  const run = getRun(workdir, runId);
101
96
  if (!run) continue;
102
97
  const runDir = runDirPath(workdir, runId);
@@ -130,7 +125,7 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
130
125
  updatedAt: checkpoint?.updatedAt ?? run.updated_at,
131
126
  });
132
127
  }
133
- // Active pointer first (priority, not exclusivity), then newest updated.
128
+ // Newest updated first; the active run (registry hint) gets priority.
134
129
  candidates.sort((a, b) => {
135
130
  const aActive = active?.run_id === a.runId ? 1 : 0;
136
131
  const bActive = active?.run_id === b.runId ? 1 : 0;
@@ -140,14 +135,17 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
140
135
  return candidates;
141
136
  }
142
137
 
143
- /** Pick the default candidate: active-unfinished first, else unique candidate (D-001). */
138
+ /**
139
+ * Pick the default candidate (v0.6.0 D-1): the SESSION BINDING is resolved by
140
+ * the command (it owns the SessionManager); here a unique candidate goes
141
+ * direct and everything else is ambiguous — the active-pointer auto-win is
142
+ * gone (multi-run workdirs must not silently pick the registry hint).
143
+ */
144
144
  export function pickDefaultCandidate(workdir: string, candidates: ResumeCandidate[]): ResumeCandidate | null {
145
+ void workdir;
145
146
  if (candidates.length === 0) return null;
146
- const active = readActive(workdir);
147
- const activeCandidate = active ? candidates.find((candidate) => candidate.runId === active.run_id) ?? null : null;
148
- if (activeCandidate !== null) return activeCandidate;
149
147
  if (candidates.length === 1) return candidates[0]!;
150
- return null; // ambiguous: the command must ask
148
+ return null; // ambiguous: the command must ask (binding first, then form)
151
149
  }
152
150
 
153
151
  export interface DecisionLedgerEntry {