pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
@@ -17,11 +17,11 @@ export const REVIEWER_LENSES: readonly ReviewerLane[] = [
17
17
  { id: "verification", lens: "verification rigor, risks, and evidence gaps" },
18
18
  ] as const;
19
19
 
20
- function buildSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
21
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
20
+ function buildSharedHeader(opts: RefinePromptInput): string {
21
+ const lensLine = opts.lens ? `\nReview lens: ${opts.lens}.` : "";
22
22
  const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
23
23
  const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
24
- return `Goal: ${role === "reviewer" ? "review the plan against the repository" : "stress-test the plan's assumptions"}.
24
+ return `Goal: review the plan against the repository and surface what needs the user's judgment.
25
25
 
26
26
  Target: ${opts.planPath}
27
27
 
@@ -30,23 +30,6 @@ Authority boundary: read-only analysis only. Do not edit, write, delete, commit,
30
30
  Evidence: inspect the repository with read, grep, find, and ls before judging the plan.${lensLine}${focusLine}${contextLine}`;
31
31
  }
32
32
 
33
- /** Shared header for the post-execution implementation review: the
34
- * accepted plan is the contract, the IMPLEMENTATION in the worktree is
35
- * under review. Findings must anchor to the plan's goals/acceptance
36
- * criteria and explicitly assess delivery maturity. */
37
- function buildImplementationSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
38
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
39
- const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
40
- const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
41
- return `Goal: ${role === "reviewer" ? "review the implemented result in the worktree against the plan" : "stress-test the implemented result's assumptions"}.
42
-
43
- Target: ${opts.planPath} (the accepted plan; the IMPLEMENTATION in the worktree is under review)
44
-
45
- Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
46
-
47
- Evidence: inspect the repository with read, grep, find, and ls before judging the implementation. Judge the implementation against the plan's goals, verifier checklist, and acceptance criteria. Explicitly assess delivery maturity: did the executor ship a minimal MVP only, or refine for long-term growth (no stopgaps, long-term architectural decisions, missing tests, technical debt, production readiness)? Calibrate severity accordingly. Out-of-scope improvement ideas are low severity by default and must not be forced into findings.${lensLine}${focusLine}${contextLine}`;
48
- }
49
-
50
33
  export function reviewerLanes(count: number): ReviewerLane[] {
51
34
  if (count === 3) return [...REVIEWER_LENSES];
52
35
  if (count === 2) return [...REVIEWER_LENSES.slice(0, 2)];
@@ -54,47 +37,22 @@ export function reviewerLanes(count: number): ReviewerLane[] {
54
37
  }
55
38
 
56
39
  export function buildReviewerTask(opts: RefinePromptInput): string {
57
- return `${buildSharedHeader("reviewer", opts)}
58
-
59
- Success criteria: return evidence-backed findings or explicitly say the plan holds up.
60
-
61
- Output: Markdown, highest severity first. For each finding use this shape:
62
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
63
-
64
- Surface at most five high-priority findings; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
65
-
66
- Plan file: ${opts.planPath}
67
-
68
- ---8<--- PLAN CONTENT ---8<---
69
- ${opts.planText}
70
- ---8<--- END PLAN CONTENT ---8<---`;
71
- }
40
+ return `${buildSharedHeader(opts)}
72
41
 
73
- export function buildCriticizerTask(opts: RefinePromptInput): string {
74
- return `${buildSharedHeader("criticizer", opts)}
42
+ Success criteria: return evidence-backed findings, plus the questions only the user can settle. If the plan holds up, say so explicitly and list what you checked.
75
43
 
76
- Success criteria: return concrete, answerable questions only; never rewrite the plan.
44
+ Output: Markdown with exactly two top-level parts, in this order.
77
45
 
78
- Output: Markdown in exactly this shape:
79
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
80
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
46
+ ## Findings
81
47
 
82
- Plan file: ${opts.planPath}
83
-
84
- ---8<--- PLAN CONTENT ---8<---
85
- ${opts.planText}
86
- ---8<--- END PLAN CONTENT ---8<---`;
87
- }
88
-
89
- export function buildImplementationReviewerTask(opts: RefinePromptInput): string {
90
- return `${buildImplementationSharedHeader("reviewer", opts)}
48
+ Highest severity first. For each finding use this shape:
49
+ - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
91
50
 
92
- Success criteria: return evidence-backed findings or explicitly say the implementation holds up.
51
+ Surface at most five high-priority findings; list lower-severity findings after them. Write "None." when there are none.
93
52
 
94
- Output: Markdown, highest severity first. For each finding use this shape:
95
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003) or files; evidence: repo path/command that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
53
+ ## Questions
96
54
 
97
- Surface at most five high-priority findings; list lower-severity findings after them. If the implementation holds up, say so explicitly and list what you checked.
55
+ At most five numbered questions (\`Q-1\`, \`Q-2\`, …) covering everything that needs the user's decision before the plan can be safely revised — hidden trade-offs, undetermined semantics, accept/reject calls on findings marked needs-discussion. Each question gets one line of why it matters, phrased so a user with repo access can answer it concretely. Never rhetorical; never questions the repository itself already answers. Stop earlier if nothing genuinely needs the user.
98
56
 
99
57
  Plan file: ${opts.planPath}
100
58
 
@@ -161,19 +119,3 @@ Section contracts:
161
119
  - Coverage: which parts of the reference you actually read versus skipped.
162
120
  - Evidence Gaps: what you could not determine from the reference alone.`;
163
121
  }
164
-
165
- export function buildImplementationCriticizerTask(opts: RefinePromptInput): string {
166
- return `${buildImplementationSharedHeader("criticizer", opts)}
167
-
168
- Success criteria: return concrete, answerable questions only; never rewrite the plan or the implementation.
169
-
170
- Output: Markdown in exactly this shape:
171
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
172
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the implementation genuinely holds.
173
-
174
- Plan file: ${opts.planPath}
175
-
176
- ---8<--- PLAN CONTENT ---8<---
177
- ${opts.planText}
178
- ---8<--- END PLAN CONTENT ---8<---`;
179
- }
@@ -10,11 +10,30 @@
10
10
  const ELLIPSIS = "…";
11
11
 
12
12
  /**
13
- * ECMA-48 CSI sequence: ESC [ parameter bytes (0x30-0x3F), intermediate bytes
14
- * (0x20-0x2F), final byte (0x40-0x7E). Matched atomically so styling payloads
15
- * never leak into width math and are never split mid-sequence.
13
+ * ECMA-48 escape sequences. ALL of them render at zero width, so styling and
14
+ * control payloads must never leak into width math nor be split mid-sequence.
15
+ *
16
+ * Three families are recognised:
17
+ * - CSI: `ESC [ params intermediates final` (SGR colours, cursor moves)
18
+ * - String-terminated: `ESC ] _ P X ^` … `BEL`|`ST` (OSC, APC, DCS, SOS, PM)
19
+ * - Simple: `ESC` intermediates `final` (`ESC c`, `ESC (B`, `ESC 7`)
20
+ *
21
+ * The string family matters beyond colour: pi's `CURSOR_MARKER` is an APC
22
+ * sequence (`ESC _ pi:c BEL`). Counting its payload as visible text made every
23
+ * row carrying the marker — e.g. a focused search Input — measure several
24
+ * columns too wide, which pushed that row's right-hand border out of
25
+ * alignment. The terminator is required: a well-formed sequence always has
26
+ * one, and refusing malformed ones keeps the fallback (the simple family)
27
+ * from swallowing real text.
16
28
  */
17
- const CSI_PATTERN = /\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]/g;
29
+ const ESCAPE_PATTERN = new RegExp(
30
+ [
31
+ "\\x1b\\[[\\x30-\\x3f]*[\\x20-\\x2f]*[\\x40-\\x7e]", // CSI
32
+ "\\x1b[\\]PX^_][^\\x07\\x1b]*(?:\\x07|\\x1b\\\\)", // OSC / APC / DCS / SOS / PM
33
+ "\\x1b[\\x20-\\x2f]*[\\x30-\\x7e]", // simple escapes
34
+ ].join("|"),
35
+ "g",
36
+ );
18
37
 
19
38
  interface AnsiPart {
20
39
  kind: "csi" | "text";
@@ -24,7 +43,7 @@ interface AnsiPart {
24
43
  function splitAnsi(text: string): AnsiPart[] {
25
44
  const parts: AnsiPart[] = [];
26
45
  let last = 0;
27
- for (const match of text.matchAll(CSI_PATTERN)) {
46
+ for (const match of text.matchAll(ESCAPE_PATTERN)) {
28
47
  const start = match.index ?? 0;
29
48
  if (start > last) parts.push({ kind: "text", value: text.slice(last, start) });
30
49
  parts.push({ kind: "csi", value: match[0] });
@@ -1,6 +1,6 @@
1
1
  import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
2
2
 
3
- export type RefineOverlayRole = "reviewer" | "criticizer" | "refs" | "executor";
3
+ export type RefineOverlayRole = "reviewer" | "refs" | "auditor";
4
4
  export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
5
5
  export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
6
6
 
package/src/refine-ui.ts CHANGED
@@ -135,10 +135,12 @@ function previewTranscriptText(entry: RefineTranscriptEntry, width: number): { l
135
135
  return { lines: lines.slice(-STREAMING_PREVIEW_LINES), truncated: true };
136
136
  }
137
137
 
138
- function footerText(laneCount: number, lang: UiLanguage): string {
138
+ function footerText(laneCount: number, lang: UiLanguage, role: RefineOverlayRole = "reviewer"): string {
139
139
  const chrome = refineChrome(lang);
140
140
  const parts = [chrome.close, chrome.scroll, chrome.page];
141
141
  if (laneCount > 1) parts.push(chrome.switchLane);
142
+ // Auditor overlay only: the reopen shortcut is the way back after ESC.
143
+ if (role === "auditor") parts.push(chrome.reopen);
142
144
  return parts.join(" · ");
143
145
  }
144
146
 
@@ -166,7 +168,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
166
168
  const complete = lanes.filter((lane) => lane.status === "complete").length;
167
169
  const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
168
170
  const running = lanes.filter((lane) => lane.status === "running").length;
169
- const title = role === "reviewer" ? "Reviewer" : role === "refs" ? "Refs" : role === "executor" ? "Executor" : "Criticizer";
171
+ const title = role === "reviewer" ? "Reviewer" : role === "auditor" ? "Execution review" : "Refs";
170
172
  const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
171
173
  const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
172
174
  return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
@@ -250,7 +252,7 @@ export class RefineOverlayComponent implements Component {
250
252
  lines.push(...this.renderPane(this.lanes[index]!, innerWidth, paneHeight, index === this.selectedLane));
251
253
  }
252
254
  }
253
- lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang)), innerWidth));
255
+ lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang, this.role)), innerWidth));
254
256
  lines.push(renderBorderLine(this.theme, innerWidth, "bottom"));
255
257
  return lines.map((line) => fitLine(line, width));
256
258
  }
@@ -352,6 +354,20 @@ export class RefineOverlayController {
352
354
  .catch(() => undefined);
353
355
  }
354
356
 
357
+ /** Repaint without a progress event (the engine mutates lane state directly). */
358
+ rerender(): void {
359
+ if (this.closed) return;
360
+ this.tui?.requestRender();
361
+ }
362
+
363
+ /** Replace the lane with a matching id with engine-held state — used by the
364
+ * execution-review reopen path so a fresh controller continues the SAME
365
+ * transcript (one-shot controllers can never re-open themselves). */
366
+ seedLane(state: RefineLaneState): void {
367
+ const index = this.lanes.findIndex((lane) => lane.id === state.id);
368
+ if (index >= 0) this.lanes[index] = state;
369
+ }
370
+
355
371
  update(laneId: string, event: SubagentProgressEvent): void {
356
372
  if (this.closed) return;
357
373
  const lane = this.lanes.find((candidate) => candidate.id === laneId);
@@ -12,14 +12,14 @@
12
12
  * - R-007: after every dialog the world is re-checked; one kickoff max.
13
13
  */
14
14
 
15
- import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
15
+ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
16
16
  import * as fs from "node:fs";
17
17
  import { existsSync } from "node:fs";
18
18
  import * as path from "node:path";
19
19
  import { loadExecutionFromCheckpoint } from "./exec.ts";
20
20
  import { bindRun, boundRunId } from "./run-context.ts";
21
21
  import { acquireOwnership, OwnershipError, releaseOwnership } from "./run-ownership.ts";
22
- import { loadConfig, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
22
+ import { loadConfig, resolveArtifactRoot, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
23
23
  import {
24
24
  applyMigration,
25
25
  mutateCheckpoint,
@@ -35,12 +35,12 @@ import {
35
35
  reconcileCheckpointWithLedger,
36
36
  type ResumeCandidate,
37
37
  } from "./resume.ts";
38
+ import { messaging } from "./messaging.ts";
38
39
 
39
40
  /** One kickoff per invocation; repeat invocations are blocked by the idle check. */
40
41
  let inFlight = false;
41
42
 
42
43
  export async function resumePlansCommand(
43
- pi: ExtensionAPI,
44
44
  ctx: ExtensionContext,
45
45
  baseDir: string,
46
46
  ): Promise<void> {
@@ -50,13 +50,13 @@ export async function resumePlansCommand(
50
50
  }
51
51
  inFlight = true;
52
52
  try {
53
- await run(pi, ctx, baseDir);
53
+ await run(ctx, baseDir);
54
54
  } finally {
55
55
  inFlight = false;
56
56
  }
57
57
  }
58
58
 
59
- async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Promise<void> {
59
+ async function run(ctx: ExtensionContext, baseDir: string): Promise<void> {
60
60
  if (!ctx.hasUI) {
61
61
  ctx.ui.notify?.("/resume-plans needs an interactive session (TUI/RPC); print/json cannot resume.", "warning");
62
62
  return;
@@ -105,7 +105,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
105
105
  }
106
106
  if (candidate.checkpointStatus === "corrupt") {
107
107
  ctx.ui.notify(
108
- `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi_plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
108
+ `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi-plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
109
109
  "error",
110
110
  );
111
111
  return;
@@ -159,7 +159,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
159
159
  }
160
160
  }
161
161
 
162
- const brief = await buildBrief(pi, ctx, baseDir, candidate);
162
+ const brief = await buildBrief(ctx, baseDir, candidate);
163
163
  if (brief === null) {
164
164
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
165
165
  return; // buildBrief reported the specific problem
@@ -167,7 +167,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
167
167
 
168
168
  // Exactly one kickoff (R-007). The idle re-check happens above and in
169
169
  // buildBrief's dialogs; the message itself starts the continuation.
170
- await pi.sendUserMessage(brief.text);
170
+ await messaging().sendUserMessage(brief.text);
171
171
  ctx.ui.notify(`Resumed ${candidate.runId} (${brief.phaseLabel}).`, "info");
172
172
  } catch (error) {
173
173
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
@@ -189,13 +189,12 @@ export function migrateRunIntoCurrentWorktree(workdir: string, candidate: Resume
189
189
  const stateRoot = loadConfigShared(workdir);
190
190
  if (stateRoot === null) return null;
191
191
  const config = loadConfig(stateRoot);
192
- let artifactRoot = config.artifact_root;
193
- if (!path.isAbsolute(artifactRoot)) artifactRoot = path.resolve(workdir, artifactRoot);
192
+ const artifactRoot = resolveArtifactRoot(workdir, config.artifact_root);
194
193
  const sourceDir = candidate.run.artifact_dir;
195
194
  if (!existsSync(sourceDir)) return null;
196
195
  // Artifacts already in a shared location stay put.
197
196
  const gitRoot = findGitCommonDir(workdir);
198
- if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi_plans"))) return sourceDir;
197
+ if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi-plans"))) return sourceDir;
199
198
  const targetDir = path.join(artifactRoot, path.basename(sourceDir));
200
199
  for (const entry of fs.readdirSync(sourceDir, { withFileTypes: true })) {
201
200
  if (!entry.isFile()) continue;
@@ -261,46 +260,7 @@ export interface ResumeBrief {
261
260
  text: string;
262
261
  }
263
262
 
264
- /** D-5 crash-window recovery: rebuild the implementation-review loop
265
- * configuration from the checkpoint plus the decisions ledger. Answers land
266
- * in the ledger first (stable questionIds), the combined record-checkpoint
267
- * second; between the two, this resolver reconstructs what is known and
268
- * reports which questions are still missing. Exported for tests. */
269
- export function resolveImplReviewConfig(
270
- review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } | undefined,
271
- ledger: { questionId?: string; answer?: string }[],
272
- ): {
273
- condition: string | undefined;
274
- conditionFromLedger: boolean;
275
- reviewerCount: number | undefined;
276
- reviewerCountFromLedger: boolean;
277
- missing: Array<"termination-condition" | "impl-review-reviewer-count">;
278
- } {
279
- const latest = (id: string): string | undefined =>
280
- [...ledger].reverse().find((entry) => entry.questionId === id)?.answer;
281
- const ledgerCondition = latest("termination-condition");
282
- const ledgerCountRaw = latest("impl-review-reviewer-count");
283
- const ledgerCount = ledgerCountRaw === undefined ? undefined : Number.parseInt(ledgerCountRaw, 10);
284
- const ledgerCountValid = ledgerCount !== undefined && Number.isInteger(ledgerCount) && ledgerCount >= 1 && ledgerCount <= 3;
285
- const condition = review?.terminationCondition ?? ledgerCondition;
286
- const reviewerCount = review?.reviewerCount ?? (ledgerCountValid ? ledgerCount : undefined);
287
- // Both questions are per-run configuration: a count is missing whenever it
288
- // is unknown (including legacy checkpoints persisted before 0.5.4),
289
- // independent of where the condition came from.
290
- const missing: Array<"termination-condition" | "impl-review-reviewer-count"> = [];
291
- if (condition === undefined) missing.push("termination-condition");
292
- if (reviewerCount === undefined) missing.push("impl-review-reviewer-count");
293
- return {
294
- condition,
295
- conditionFromLedger: review?.terminationCondition === undefined && ledgerCondition !== undefined,
296
- reviewerCount,
297
- reviewerCountFromLedger: review?.reviewerCount === undefined && ledgerCountValid,
298
- missing,
299
- };
300
- }
301
-
302
263
  async function buildBrief(
303
- pi: ExtensionAPI,
304
264
  ctx: ExtensionContext,
305
265
  baseDir: string,
306
266
  candidate: ResumeCandidate,
@@ -327,7 +287,7 @@ async function buildBrief(
327
287
  bindRun(ctx.sessionManager, ctx.cwd, runId);
328
288
 
329
289
  if (cp.phase === "executing") {
330
- const load = loadExecutionFromCheckpoint(pi, ctx, runId);
290
+ const load = loadExecutionFromCheckpoint(ctx, runId);
331
291
  if (load.status === "loaded") {
332
292
  if (run.status === "stopped" || run.status === "accepted") {
333
293
  try {
@@ -338,14 +298,32 @@ async function buildBrief(
338
298
  }
339
299
  const doneList = (load.doneVcIds ?? []).join(", ") || "none";
340
300
  const reverify = load.reverifyAll
341
- ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously verified VC must be re-verified before new work counts. Historically verified (evidence only): ${doneList}.`
301
+ ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
342
302
  : `\nPreviously verified and still valid: ${doneList}.`;
343
- const paused = load.pausedReason ? `\nExecution was paused: ${load.pausedReason}. Continue from where it stopped.` : "";
303
+ // v0.8: a review-cap pause is NOT cleared by this resume — only
304
+ // /plans-execute (an explicit user confirmation) grants a fresh
305
+ // five-round budget; ordinary resumes and input keep it paused.
306
+ const paused = load.pausedReason
307
+ ? load.pausedReason.startsWith("execution review exhausted") || load.pausedReason.startsWith("completion audit exhausted")
308
+ ? `\nExecution had been paused: ${load.pausedReason} — this pause survives the resume; run /plans-execute to grant a fresh five-round review budget.`
309
+ : `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.`
310
+ : "";
311
+ const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
312
+ // v0.8: a verifying run keeps checkpoint phase "executing" but the run
313
+ // STATUS is verifying — surface which loop owns the run right now.
314
+ const verifying = run.status === "verifying";
344
315
  return {
345
- phaseLabel: "executing",
346
- text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}\nFollow the execution-loop contract: implement in dependency order, verify each VC, and mark completions with [DONE:VC-xxx]. The remaining checklist is injected each turn.`,
316
+ phaseLabel: verifying ? "verifying" : "executing",
317
+ text: `[PI-PLANS RESUME] ${verifying ? "Execution review of" : "Execution of"} run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\n${verifying ? "The task tree is terminal and the execution-review loop owns the run: when all tasks are terminal and checks are still owed, a read-only reviewer round runs automatically (status verifying → done when every check passes). If a check fails, its tasks roll back to pending — fix and re-close them with plans_update_task." : "Follow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the execution reviewer verify the checks. The current wave and remaining tasks are injected each turn."}`,
347
318
  };
348
319
  }
320
+ if (load.legacyDelegate) {
321
+ ctx.ui.notify(
322
+ `Cannot resume ${runId} directly: it was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Run /plans-execute to re-approve the handoff; execution restarts from the first task (0.6.0 progress cannot map onto the task tree).`,
323
+ "warning",
324
+ );
325
+ return null;
326
+ }
349
327
  if (load.status === "plan-missing" || load.status === "plan-mismatch") {
350
328
  ctx.ui.notify(`Cannot resume ${runId}: ${load.error ?? "plan file missing"}.`, "error");
351
329
  return null;
@@ -359,69 +337,19 @@ async function buildBrief(
359
337
  }
360
338
 
361
339
  if (cp.phase === "implementation-review") {
362
- const review = cp.implementationReview;
363
- const condition = review?.terminationCondition;
364
- const lines: string[] = [
365
- `[PI-PLANS RESUME] Implementation review of run ${runId} continues in this session.`,
366
- `Plan: ${cp.plan?.path ?? "(unknown)"}`,
367
- ];
368
- // D-5 crash-window recovery: both config answers live in the decisions
369
- // ledger under stable questionIds. Rebuild from the ledger, ask ONLY the
370
- // missing question(s), then persist BOTH in one combined write. The
371
- // re-ask guard on the checkpoint (termination condition already
372
- // configured) rejects duplicate writes, so the combined write happens
373
- // only while the field is genuinely absent.
374
- const ledger = readDecisionLedger(ctx.cwd, runId);
375
- const resolved = resolveImplReviewConfig(review, ledger);
376
- const effectiveCondition = resolved.condition;
377
- const effectiveCount = resolved.reviewerCount;
378
- if (effectiveCondition === undefined) {
379
- lines.push(
380
- `The termination condition was never chosen. Ask it now via ask_choice (autoComplete: false, questionId: "termination-condition"): "How should the implementation-review loop terminate?" Options: goal wait (recommended) / until no high-severity finding (hard cap 5 rounds) / 1 / 2 / 3 rounds.`,
381
- );
382
- } else {
383
- lines.push(`Termination condition: ${effectiveCondition}${resolved.conditionFromLedger ? " (recovered from the decisions ledger)" : ""}`);
384
- }
385
- if (review?.reviewerCount === undefined) {
386
- if (effectiveCount !== undefined && Number.isInteger(effectiveCount) && effectiveCount >= 1 && effectiveCount <= 3) {
387
- lines.push(`Reviewer count: ${effectiveCount} (recovered from the decisions ledger; not yet persisted).`);
388
- } else {
389
- lines.push(
390
- `The reviewer count was never chosen. Ask it via ask_choice (autoComplete: false, allowOther: false, questionId: "impl-review-reviewer-count", digit labels "1"/"2"/"3"): "How many concurrent reviewers should each implementation-review round use?" — recommended default follows this run's skill: plan-big / plan-with-refs → 3, others → 1.`,
391
- );
392
- }
393
- } else {
394
- lines.push(`Reviewers per round: ${review.reviewerCount}`);
395
- }
396
- if (condition === undefined) {
397
- // Only when the checkpoint is still unconfigured: the combined write
398
- // (both answers) is one transaction. When only terminationCondition is
399
- // missing but a ledger answer exists for both, write both; when a
400
- // question is genuinely unanswered, ask first, then write.
401
- lines.push(
402
- `After both answers exist (asked now or recovered from the ledger), persist them in ONE call: plans record-checkpoint (checkpoint: { transition: "implementation-review-configured", terminationCondition: "<answer>", reviewerCount: <integer> }).`,
403
- );
404
- } else {
405
- lines.push(`Completed rounds in this worktree: ${review?.completedRounds ?? 0} (hard cap 5).`);
406
- }
407
- const currentRound = review?.currentRoundId
408
- ? cp.reviewRounds.find((round) => round.roundId === review.currentRoundId)
409
- : undefined;
410
- if (currentRound) {
411
- const done = currentRound.lanes.filter((lane) => lane.status === "complete").map((lane) => lane.laneId);
412
- const pending = currentRound.lanes.filter((lane) => lane.status !== "complete").map((lane) => lane.laneId);
413
- lines.push(
414
- `Round ${currentRound.roundId} is in flight — complete lanes: ${done.join(", ") || "none"}; pending/failed lanes: ${pending.join(", ") || "none"}. Resume it with refine (role: "reviewer", target: "implementation", resumeRoundId: "${currentRound.roundId}") so completed lanes are reused, never re-run.`,
415
- );
416
- } else {
417
- lines.push(
418
- `Start the next round with refine (role: "reviewer", target: "implementation", reviewers: ${review?.reviewerCount ?? effectiveCount ?? "<configured count>"}) — do NOT pass a resumeRoundId unless resuming an interrupted round; an omitted reviewers argument falls back to the run's configured reviewerCount.`,
419
- );
340
+ // v0.6.1 (D-018/D-020): the implementation-review loop is gone; a
341
+ // legacy 0.6.0 checkpoint in this phase maps to execution completed.
342
+ // The run flips to done so the status line and run registry agree.
343
+ try {
344
+ setRunStatus(ctx.cwd, runId, "done");
345
+ } catch {
346
+ /* best-effort */
420
347
  }
421
- lines.push(
422
- `Record boundaries with plans record-checkpoint: review-consolidated → implementation-round-finished per round; completed (with evidence) when the termination condition is met.`,
348
+ ctx.ui.notify(
349
+ `${runId} finished under the removed v0.6.0 implementation-review loop; mapped to done. Its historical acceptance stands; no review loop to resume.`,
350
+ "info",
423
351
  );
424
- return { phaseLabel: "implementation-review", text: lines.join("\n") };
352
+ return null;
425
353
  }
426
354
 
427
355
  // planning / reviewing: rebuild the workflow context (R-002/R-003).
@@ -512,19 +440,7 @@ async function buildLegacyBrief(
512
440
  `This run predates durable checkpoints: treat every VC as unverified and re-run the execution handoff (execute_plan or /plans-execute) for explicit approval before writing any code.`,
513
441
  );
514
442
  }
515
- if (candidate.run.status === "done") {
516
- const ok = await ctx.ui.confirm(
517
- "Finished run with review artifacts",
518
- `${candidate.runId} is marked done and has review records, but completion of the implementation-review loop cannot be proven for legacy runs. Resume the review loop anyway?`,
519
- );
520
- if (!ok) {
521
- ctx.ui.notify("Cancelled; nothing was changed.", "info");
522
- return null;
523
- }
524
- lines.push(
525
- `Resume the implementation-review loop: ask the termination condition (questionId: "termination-condition") if unknown, then run rounds with refine (target: "implementation").`,
526
- );
527
- }
528
443
  lines.push(`Continue only the missing work; never restart the interview from scratch.`);
529
444
  return { phaseLabel: candidate.phaseLabel, text: lines.join("\n") };
530
445
  }
446
+
package/src/resume.ts CHANGED
@@ -27,7 +27,7 @@ export interface ResumeCandidate {
27
27
  updatedAt: string;
28
28
  }
29
29
 
30
- const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "stopped"]);
30
+ const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "verifying", "stopped"]);
31
31
 
32
32
  /** Terminal runs are resumable only when unfinished implementation-review evidence exists (D-008). */
33
33
  function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null): boolean {
@@ -52,6 +52,10 @@ function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null
52
52
  }
53
53
 
54
54
  function phaseLabelOf(run: RunInfo, checkpoint: WorkflowCheckpoint | null): string {
55
+ // A verifying run keeps checkpoint.phase === "executing" (the execution
56
+ // state machine's phase is unchanged); surface the run status instead so
57
+ // the resume list never hides the verification loop behind "executing".
58
+ if (run.status === "verifying") return "verifying";
55
59
  if (checkpoint !== null) return checkpoint.phase;
56
60
  if (run.status === "stopped") return "executing (stopped)";
57
61
  return run.status;