pi-plans 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (109) hide show
  1. package/CONTRIBUTING.md +3 -3
  2. package/README.md +39 -37
  3. package/agents/reviewer.md +12 -3
  4. package/index.ts +42 -35
  5. package/package.json +1 -1
  6. package/references/pi-planning-workflow.md +44 -60
  7. package/references/plan-artifact-template.md +71 -60
  8. package/references/state-and-config.md +59 -43
  9. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  10. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  11. package/scripts/run-tests.ts +12 -1
  12. package/scripts/validate.ts +20 -9
  13. package/skills/debug-and-plan/SKILL.md +3 -3
  14. package/skills/plan-big/SKILL.md +3 -3
  15. package/skills/plan-normal/SKILL.md +3 -3
  16. package/skills/plan-small/SKILL.md +4 -4
  17. package/skills/plan-with-refs/SKILL.md +6 -6
  18. package/skills/planning/SKILL.md +1 -1
  19. package/src/ask-form.ts +4 -4
  20. package/src/auditor.ts +126 -0
  21. package/src/auto-approve.ts +1 -1
  22. package/src/autocomplete.ts +19 -17
  23. package/src/code-graph/commands.ts +2 -2
  24. package/src/code-graph/community.ts +1 -1
  25. package/src/code-graph/paths.ts +1 -1
  26. package/src/code-graph/watch.ts +2 -2
  27. package/src/compaction.ts +3 -3
  28. package/src/config-command.ts +146 -73
  29. package/src/dashboard.ts +257 -0
  30. package/src/exec.ts +692 -919
  31. package/src/global-state.ts +304 -0
  32. package/src/guard.ts +18 -19
  33. package/src/messaging.ts +44 -0
  34. package/src/plan.ts +421 -112
  35. package/src/query-hook.ts +4 -4
  36. package/src/refine-prompts.ts +12 -70
  37. package/src/refine-ui-helpers.ts +24 -5
  38. package/src/refine-ui-state.ts +1 -1
  39. package/src/refine-ui.ts +1 -1
  40. package/src/resume-command.ts +34 -128
  41. package/src/role-panels.ts +542 -0
  42. package/src/run-context.ts +3 -10
  43. package/src/state.ts +272 -72
  44. package/src/subagent.ts +19 -29
  45. package/src/task-tool.ts +100 -0
  46. package/src/tasks.ts +189 -0
  47. package/src/thinking-levels.ts +67 -0
  48. package/src/ui-language.ts +3 -54
  49. package/src/workflow-state.ts +63 -58
  50. package/tests/analyze-refs.test.ts +35 -18
  51. package/tests/ask-choice-schema.test.ts +0 -12
  52. package/tests/ask-choice.test.ts +2 -49
  53. package/tests/ask-form-tool.test.ts +4 -5
  54. package/tests/ask-form.test.ts +2 -2
  55. package/tests/auditor.test.ts +111 -0
  56. package/tests/auto-approve.test.ts +7 -10
  57. package/tests/autocomplete.test.ts +8 -11
  58. package/tests/code-graph-apply-action.test.ts +2 -2
  59. package/tests/code-graph-commands.test.ts +2 -2
  60. package/tests/code-graph-index.test.ts +2 -2
  61. package/tests/code-graph-loop.e2e.test.ts +1 -1
  62. package/tests/code-graph-mutations.test.ts +1 -1
  63. package/tests/code-graph-rollback.test.ts +1 -1
  64. package/tests/code-graph-v05.test.ts +2 -2
  65. package/tests/compaction.test.ts +1 -1
  66. package/tests/config-command.test.ts +103 -100
  67. package/tests/dashboard.test.ts +268 -0
  68. package/tests/exec-lifecycle.test.ts +181 -115
  69. package/tests/exec-panel-lifecycle.test.ts +106 -251
  70. package/tests/exec.test.ts +617 -1706
  71. package/tests/execute-plan.test.ts +44 -19
  72. package/tests/extension-load.test.ts +48 -0
  73. package/tests/global-state.test.ts +371 -0
  74. package/tests/graph-aware-file-tools.test.ts +5 -5
  75. package/tests/guard.test.ts +1 -1
  76. package/tests/multi-run.test.ts +3 -103
  77. package/tests/plan.test.ts +139 -62
  78. package/tests/plans.test.ts +7 -79
  79. package/tests/refine-prompts.test.ts +20 -71
  80. package/tests/refine-resume.test.ts +27 -22
  81. package/tests/refine-ui.test.ts +6 -15
  82. package/tests/resume-lifecycle.test.ts +37 -22
  83. package/tests/resume.test.ts +33 -81
  84. package/tests/role-panels.test.ts +391 -0
  85. package/tests/run-context.test.ts +1 -1
  86. package/tests/run-ownership.test.ts +1 -1
  87. package/tests/stale-ctx.test.ts +218 -0
  88. package/tests/state.test.ts +151 -32
  89. package/tests/subagent-thinking.test.ts +65 -0
  90. package/tests/subagent-usage.test.ts +1 -1
  91. package/tests/task-tool.test.ts +61 -0
  92. package/tests/thinking-levels.test.ts +77 -0
  93. package/tests/ui-language.test.ts +2 -17
  94. package/tests/workflow-state.test.ts +17 -99
  95. package/tools/analyze-refs.ts +67 -32
  96. package/tools/ask-choice.ts +7 -53
  97. package/tools/code-graph.ts +2 -2
  98. package/tools/execute-plan.ts +48 -99
  99. package/tools/graph-aware-file-tools.ts +4 -10
  100. package/tools/plans.ts +40 -66
  101. package/tools/refine.ts +101 -164
  102. package/agents/criticizer.md +0 -18
  103. package/agents/executor.md +0 -26
  104. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  105. package/src/panel.ts +0 -473
  106. package/src/termination-prompt.ts +0 -73
  107. package/tests/goal-wait.test.ts +0 -269
  108. package/tests/panel-i-zero.test.ts +0 -420
  109. package/tests/panel.test.ts +0 -355
@@ -17,11 +17,11 @@ export const REVIEWER_LENSES: readonly ReviewerLane[] = [
17
17
  { id: "verification", lens: "verification rigor, risks, and evidence gaps" },
18
18
  ] as const;
19
19
 
20
- function buildSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
21
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
20
+ function buildSharedHeader(opts: RefinePromptInput): string {
21
+ const lensLine = opts.lens ? `\nReview lens: ${opts.lens}.` : "";
22
22
  const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
23
23
  const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
24
- return `Goal: ${role === "reviewer" ? "review the plan against the repository" : "stress-test the plan's assumptions"}.
24
+ return `Goal: review the plan against the repository and surface what needs the user's judgment.
25
25
 
26
26
  Target: ${opts.planPath}
27
27
 
@@ -30,23 +30,6 @@ Authority boundary: read-only analysis only. Do not edit, write, delete, commit,
30
30
  Evidence: inspect the repository with read, grep, find, and ls before judging the plan.${lensLine}${focusLine}${contextLine}`;
31
31
  }
32
32
 
33
- /** Shared header for the post-execution implementation review: the
34
- * accepted plan is the contract, the IMPLEMENTATION in the worktree is
35
- * under review. Findings must anchor to the plan's goals/acceptance
36
- * criteria and explicitly assess delivery maturity. */
37
- function buildImplementationSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
38
- const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
39
- const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
40
- const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
41
- return `Goal: ${role === "reviewer" ? "review the implemented result in the worktree against the plan" : "stress-test the implemented result's assumptions"}.
42
-
43
- Target: ${opts.planPath} (the accepted plan; the IMPLEMENTATION in the worktree is under review)
44
-
45
- Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
46
-
47
- Evidence: inspect the repository with read, grep, find, and ls before judging the implementation. Judge the implementation against the plan's goals, verifier checklist, and acceptance criteria. Explicitly assess delivery maturity: did the executor ship a minimal MVP only, or refine for long-term growth (no stopgaps, long-term architectural decisions, missing tests, technical debt, production readiness)? Calibrate severity accordingly. Out-of-scope improvement ideas are low severity by default and must not be forced into findings.${lensLine}${focusLine}${contextLine}`;
48
- }
49
-
50
33
  export function reviewerLanes(count: number): ReviewerLane[] {
51
34
  if (count === 3) return [...REVIEWER_LENSES];
52
35
  if (count === 2) return [...REVIEWER_LENSES.slice(0, 2)];
@@ -54,47 +37,22 @@ export function reviewerLanes(count: number): ReviewerLane[] {
54
37
  }
55
38
 
56
39
  export function buildReviewerTask(opts: RefinePromptInput): string {
57
- return `${buildSharedHeader("reviewer", opts)}
58
-
59
- Success criteria: return evidence-backed findings or explicitly say the plan holds up.
60
-
61
- Output: Markdown, highest severity first. For each finding use this shape:
62
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
63
-
64
- Surface at most five high-priority findings; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
65
-
66
- Plan file: ${opts.planPath}
67
-
68
- ---8<--- PLAN CONTENT ---8<---
69
- ${opts.planText}
70
- ---8<--- END PLAN CONTENT ---8<---`;
71
- }
40
+ return `${buildSharedHeader(opts)}
72
41
 
73
- export function buildCriticizerTask(opts: RefinePromptInput): string {
74
- return `${buildSharedHeader("criticizer", opts)}
42
+ Success criteria: return evidence-backed findings, plus the questions only the user can settle. If the plan holds up, say so explicitly and list what you checked.
75
43
 
76
- Success criteria: return concrete, answerable questions only; never rewrite the plan.
44
+ Output: Markdown with exactly two top-level parts, in this order.
77
45
 
78
- Output: Markdown in exactly this shape:
79
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
80
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
46
+ ## Findings
81
47
 
82
- Plan file: ${opts.planPath}
83
-
84
- ---8<--- PLAN CONTENT ---8<---
85
- ${opts.planText}
86
- ---8<--- END PLAN CONTENT ---8<---`;
87
- }
88
-
89
- export function buildImplementationReviewerTask(opts: RefinePromptInput): string {
90
- return `${buildImplementationSharedHeader("reviewer", opts)}
48
+ Highest severity first. For each finding use this shape:
49
+ - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
91
50
 
92
- Success criteria: return evidence-backed findings or explicitly say the implementation holds up.
51
+ Surface at most five high-priority findings; list lower-severity findings after them. Write "None." when there are none.
93
52
 
94
- Output: Markdown, highest severity first. For each finding use this shape:
95
- - \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003) or files; evidence: repo path/command that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
53
+ ## Questions
96
54
 
97
- Surface at most five high-priority findings; list lower-severity findings after them. If the implementation holds up, say so explicitly and list what you checked.
55
+ At most five numbered questions (\`Q-1\`, \`Q-2\`, …) covering everything that needs the user's decision before the plan can be safely revised — hidden trade-offs, undetermined semantics, accept/reject calls on findings marked needs-discussion. Each question gets one line of why it matters, phrased so a user with repo access can answer it concretely. Never rhetorical; never questions the repository itself already answers. Stop earlier if nothing genuinely needs the user.
98
56
 
99
57
  Plan file: ${opts.planPath}
100
58
 
@@ -161,19 +119,3 @@ Section contracts:
161
119
  - Coverage: which parts of the reference you actually read versus skipped.
162
120
  - Evidence Gaps: what you could not determine from the reference alone.`;
163
121
  }
164
-
165
- export function buildImplementationCriticizerTask(opts: RefinePromptInput): string {
166
- return `${buildImplementationSharedHeader("criticizer", opts)}
167
-
168
- Success criteria: return concrete, answerable questions only; never rewrite the plan or the implementation.
169
-
170
- Output: Markdown in exactly this shape:
171
- 1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
172
- 2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the implementation genuinely holds.
173
-
174
- Plan file: ${opts.planPath}
175
-
176
- ---8<--- PLAN CONTENT ---8<---
177
- ${opts.planText}
178
- ---8<--- END PLAN CONTENT ---8<---`;
179
- }
@@ -10,11 +10,30 @@
10
10
  const ELLIPSIS = "…";
11
11
 
12
12
  /**
13
- * ECMA-48 CSI sequence: ESC [ parameter bytes (0x30-0x3F), intermediate bytes
14
- * (0x20-0x2F), final byte (0x40-0x7E). Matched atomically so styling payloads
15
- * never leak into width math and are never split mid-sequence.
13
+ * ECMA-48 escape sequences. ALL of them render at zero width, so styling and
14
+ * control payloads must never leak into width math nor be split mid-sequence.
15
+ *
16
+ * Three families are recognised:
17
+ * - CSI: `ESC [ params intermediates final` (SGR colours, cursor moves)
18
+ * - String-terminated: `ESC ] _ P X ^` … `BEL`|`ST` (OSC, APC, DCS, SOS, PM)
19
+ * - Simple: `ESC` intermediates `final` (`ESC c`, `ESC (B`, `ESC 7`)
20
+ *
21
+ * The string family matters beyond colour: pi's `CURSOR_MARKER` is an APC
22
+ * sequence (`ESC _ pi:c BEL`). Counting its payload as visible text made every
23
+ * row carrying the marker — e.g. a focused search Input — measure several
24
+ * columns too wide, which pushed that row's right-hand border out of
25
+ * alignment. The terminator is required: a well-formed sequence always has
26
+ * one, and refusing malformed ones keeps the fallback (the simple family)
27
+ * from swallowing real text.
16
28
  */
17
- const CSI_PATTERN = /\x1b\[[\x30-\x3f]*[\x20-\x2f]*[\x40-\x7e]/g;
29
+ const ESCAPE_PATTERN = new RegExp(
30
+ [
31
+ "\\x1b\\[[\\x30-\\x3f]*[\\x20-\\x2f]*[\\x40-\\x7e]", // CSI
32
+ "\\x1b[\\]PX^_][^\\x07\\x1b]*(?:\\x07|\\x1b\\\\)", // OSC / APC / DCS / SOS / PM
33
+ "\\x1b[\\x20-\\x2f]*[\\x30-\\x7e]", // simple escapes
34
+ ].join("|"),
35
+ "g",
36
+ );
18
37
 
19
38
  interface AnsiPart {
20
39
  kind: "csi" | "text";
@@ -24,7 +43,7 @@ interface AnsiPart {
24
43
  function splitAnsi(text: string): AnsiPart[] {
25
44
  const parts: AnsiPart[] = [];
26
45
  let last = 0;
27
- for (const match of text.matchAll(CSI_PATTERN)) {
46
+ for (const match of text.matchAll(ESCAPE_PATTERN)) {
28
47
  const start = match.index ?? 0;
29
48
  if (start > last) parts.push({ kind: "text", value: text.slice(last, start) });
30
49
  parts.push({ kind: "csi", value: match[0] });
@@ -1,6 +1,6 @@
1
1
  import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
2
2
 
3
- export type RefineOverlayRole = "reviewer" | "criticizer" | "refs" | "executor";
3
+ export type RefineOverlayRole = "reviewer" | "refs";
4
4
  export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
5
5
  export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
6
6
 
package/src/refine-ui.ts CHANGED
@@ -166,7 +166,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
166
166
  const complete = lanes.filter((lane) => lane.status === "complete").length;
167
167
  const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
168
168
  const running = lanes.filter((lane) => lane.status === "running").length;
169
- const title = role === "reviewer" ? "Reviewer" : role === "refs" ? "Refs" : role === "executor" ? "Executor" : "Criticizer";
169
+ const title = role === "reviewer" ? "Reviewer" : "Refs";
170
170
  const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
171
171
  const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
172
172
  return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
@@ -12,14 +12,14 @@
12
12
  * - R-007: after every dialog the world is re-checked; one kickoff max.
13
13
  */
14
14
 
15
- import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
15
+ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
16
16
  import * as fs from "node:fs";
17
17
  import { existsSync } from "node:fs";
18
18
  import * as path from "node:path";
19
19
  import { loadExecutionFromCheckpoint } from "./exec.ts";
20
20
  import { bindRun, boundRunId } from "./run-context.ts";
21
21
  import { acquireOwnership, OwnershipError, releaseOwnership } from "./run-ownership.ts";
22
- import { loadConfig, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
22
+ import { loadConfig, resolveArtifactRoot, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
23
23
  import {
24
24
  applyMigration,
25
25
  mutateCheckpoint,
@@ -35,12 +35,12 @@ import {
35
35
  reconcileCheckpointWithLedger,
36
36
  type ResumeCandidate,
37
37
  } from "./resume.ts";
38
+ import { messaging } from "./messaging.ts";
38
39
 
39
40
  /** One kickoff per invocation; repeat invocations are blocked by the idle check. */
40
41
  let inFlight = false;
41
42
 
42
43
  export async function resumePlansCommand(
43
- pi: ExtensionAPI,
44
44
  ctx: ExtensionContext,
45
45
  baseDir: string,
46
46
  ): Promise<void> {
@@ -50,13 +50,13 @@ export async function resumePlansCommand(
50
50
  }
51
51
  inFlight = true;
52
52
  try {
53
- await run(pi, ctx, baseDir);
53
+ await run(ctx, baseDir);
54
54
  } finally {
55
55
  inFlight = false;
56
56
  }
57
57
  }
58
58
 
59
- async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Promise<void> {
59
+ async function run(ctx: ExtensionContext, baseDir: string): Promise<void> {
60
60
  if (!ctx.hasUI) {
61
61
  ctx.ui.notify?.("/resume-plans needs an interactive session (TUI/RPC); print/json cannot resume.", "warning");
62
62
  return;
@@ -105,7 +105,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
105
105
  }
106
106
  if (candidate.checkpointStatus === "corrupt") {
107
107
  ctx.ui.notify(
108
- `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi_plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
108
+ `Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi-plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
109
109
  "error",
110
110
  );
111
111
  return;
@@ -159,7 +159,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
159
159
  }
160
160
  }
161
161
 
162
- const brief = await buildBrief(pi, ctx, baseDir, candidate);
162
+ const brief = await buildBrief(ctx, baseDir, candidate);
163
163
  if (brief === null) {
164
164
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
165
165
  return; // buildBrief reported the specific problem
@@ -167,7 +167,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
167
167
 
168
168
  // Exactly one kickoff (R-007). The idle re-check happens above and in
169
169
  // buildBrief's dialogs; the message itself starts the continuation.
170
- await pi.sendUserMessage(brief.text);
170
+ await messaging().sendUserMessage(brief.text);
171
171
  ctx.ui.notify(`Resumed ${candidate.runId} (${brief.phaseLabel}).`, "info");
172
172
  } catch (error) {
173
173
  releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
@@ -189,13 +189,12 @@ export function migrateRunIntoCurrentWorktree(workdir: string, candidate: Resume
189
189
  const stateRoot = loadConfigShared(workdir);
190
190
  if (stateRoot === null) return null;
191
191
  const config = loadConfig(stateRoot);
192
- let artifactRoot = config.artifact_root;
193
- if (!path.isAbsolute(artifactRoot)) artifactRoot = path.resolve(workdir, artifactRoot);
192
+ const artifactRoot = resolveArtifactRoot(workdir, config.artifact_root);
194
193
  const sourceDir = candidate.run.artifact_dir;
195
194
  if (!existsSync(sourceDir)) return null;
196
195
  // Artifacts already in a shared location stay put.
197
196
  const gitRoot = findGitCommonDir(workdir);
198
- if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi_plans"))) return sourceDir;
197
+ if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi-plans"))) return sourceDir;
199
198
  const targetDir = path.join(artifactRoot, path.basename(sourceDir));
200
199
  for (const entry of fs.readdirSync(sourceDir, { withFileTypes: true })) {
201
200
  if (!entry.isFile()) continue;
@@ -261,46 +260,7 @@ export interface ResumeBrief {
261
260
  text: string;
262
261
  }
263
262
 
264
- /** D-5 crash-window recovery: rebuild the implementation-review loop
265
- * configuration from the checkpoint plus the decisions ledger. Answers land
266
- * in the ledger first (stable questionIds), the combined record-checkpoint
267
- * second; between the two, this resolver reconstructs what is known and
268
- * reports which questions are still missing. Exported for tests. */
269
- export function resolveImplReviewConfig(
270
- review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } | undefined,
271
- ledger: { questionId?: string; answer?: string }[],
272
- ): {
273
- condition: string | undefined;
274
- conditionFromLedger: boolean;
275
- reviewerCount: number | undefined;
276
- reviewerCountFromLedger: boolean;
277
- missing: Array<"termination-condition" | "impl-review-reviewer-count">;
278
- } {
279
- const latest = (id: string): string | undefined =>
280
- [...ledger].reverse().find((entry) => entry.questionId === id)?.answer;
281
- const ledgerCondition = latest("termination-condition");
282
- const ledgerCountRaw = latest("impl-review-reviewer-count");
283
- const ledgerCount = ledgerCountRaw === undefined ? undefined : Number.parseInt(ledgerCountRaw, 10);
284
- const ledgerCountValid = ledgerCount !== undefined && Number.isInteger(ledgerCount) && ledgerCount >= 1 && ledgerCount <= 3;
285
- const condition = review?.terminationCondition ?? ledgerCondition;
286
- const reviewerCount = review?.reviewerCount ?? (ledgerCountValid ? ledgerCount : undefined);
287
- // Both questions are per-run configuration: a count is missing whenever it
288
- // is unknown (including legacy checkpoints persisted before 0.5.4),
289
- // independent of where the condition came from.
290
- const missing: Array<"termination-condition" | "impl-review-reviewer-count"> = [];
291
- if (condition === undefined) missing.push("termination-condition");
292
- if (reviewerCount === undefined) missing.push("impl-review-reviewer-count");
293
- return {
294
- condition,
295
- conditionFromLedger: review?.terminationCondition === undefined && ledgerCondition !== undefined,
296
- reviewerCount,
297
- reviewerCountFromLedger: review?.reviewerCount === undefined && ledgerCountValid,
298
- missing,
299
- };
300
- }
301
-
302
263
  async function buildBrief(
303
- pi: ExtensionAPI,
304
264
  ctx: ExtensionContext,
305
265
  baseDir: string,
306
266
  candidate: ResumeCandidate,
@@ -327,7 +287,7 @@ async function buildBrief(
327
287
  bindRun(ctx.sessionManager, ctx.cwd, runId);
328
288
 
329
289
  if (cp.phase === "executing") {
330
- const load = loadExecutionFromCheckpoint(pi, ctx, runId);
290
+ const load = loadExecutionFromCheckpoint(ctx, runId);
331
291
  if (load.status === "loaded") {
332
292
  if (run.status === "stopped" || run.status === "accepted") {
333
293
  try {
@@ -338,14 +298,22 @@ async function buildBrief(
338
298
  }
339
299
  const doneList = (load.doneVcIds ?? []).join(", ") || "none";
340
300
  const reverify = load.reverifyAll
341
- ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously verified VC must be re-verified before new work counts. Historically verified (evidence only): ${doneList}.`
301
+ ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
342
302
  : `\nPreviously verified and still valid: ${doneList}.`;
343
- const paused = load.pausedReason ? `\nExecution was paused: ${load.pausedReason}. Continue from where it stopped.` : "";
303
+ const paused = load.pausedReason ? `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.` : "";
304
+ const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
344
305
  return {
345
306
  phaseLabel: "executing",
346
- text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}\nFollow the execution-loop contract: implement in dependency order, verify each VC, and mark completions with [DONE:VC-xxx]. The remaining checklist is injected each turn.`,
307
+ text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\nFollow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the completion auditor verify the checks. The current wave and remaining tasks are injected each turn.`,
347
308
  };
348
309
  }
310
+ if (load.legacyDelegate) {
311
+ ctx.ui.notify(
312
+ `Cannot resume ${runId} directly: it was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Run /plans-execute to re-approve the handoff; execution restarts from the first task (0.6.0 progress cannot map onto the task tree).`,
313
+ "warning",
314
+ );
315
+ return null;
316
+ }
349
317
  if (load.status === "plan-missing" || load.status === "plan-mismatch") {
350
318
  ctx.ui.notify(`Cannot resume ${runId}: ${load.error ?? "plan file missing"}.`, "error");
351
319
  return null;
@@ -359,69 +327,19 @@ async function buildBrief(
359
327
  }
360
328
 
361
329
  if (cp.phase === "implementation-review") {
362
- const review = cp.implementationReview;
363
- const condition = review?.terminationCondition;
364
- const lines: string[] = [
365
- `[PI-PLANS RESUME] Implementation review of run ${runId} continues in this session.`,
366
- `Plan: ${cp.plan?.path ?? "(unknown)"}`,
367
- ];
368
- // D-5 crash-window recovery: both config answers live in the decisions
369
- // ledger under stable questionIds. Rebuild from the ledger, ask ONLY the
370
- // missing question(s), then persist BOTH in one combined write. The
371
- // re-ask guard on the checkpoint (termination condition already
372
- // configured) rejects duplicate writes, so the combined write happens
373
- // only while the field is genuinely absent.
374
- const ledger = readDecisionLedger(ctx.cwd, runId);
375
- const resolved = resolveImplReviewConfig(review, ledger);
376
- const effectiveCondition = resolved.condition;
377
- const effectiveCount = resolved.reviewerCount;
378
- if (effectiveCondition === undefined) {
379
- lines.push(
380
- `The termination condition was never chosen. Ask it now via ask_choice (autoComplete: false, questionId: "termination-condition"): "How should the implementation-review loop terminate?" Options: goal wait (recommended) / until no high-severity finding (hard cap 5 rounds) / 1 / 2 / 3 rounds.`,
381
- );
382
- } else {
383
- lines.push(`Termination condition: ${effectiveCondition}${resolved.conditionFromLedger ? " (recovered from the decisions ledger)" : ""}`);
384
- }
385
- if (review?.reviewerCount === undefined) {
386
- if (effectiveCount !== undefined && Number.isInteger(effectiveCount) && effectiveCount >= 1 && effectiveCount <= 3) {
387
- lines.push(`Reviewer count: ${effectiveCount} (recovered from the decisions ledger; not yet persisted).`);
388
- } else {
389
- lines.push(
390
- `The reviewer count was never chosen. Ask it via ask_choice (autoComplete: false, allowOther: false, questionId: "impl-review-reviewer-count", digit labels "1"/"2"/"3"): "How many concurrent reviewers should each implementation-review round use?" — recommended default follows this run's skill: plan-big / plan-with-refs → 3, others → 1.`,
391
- );
392
- }
393
- } else {
394
- lines.push(`Reviewers per round: ${review.reviewerCount}`);
395
- }
396
- if (condition === undefined) {
397
- // Only when the checkpoint is still unconfigured: the combined write
398
- // (both answers) is one transaction. When only terminationCondition is
399
- // missing but a ledger answer exists for both, write both; when a
400
- // question is genuinely unanswered, ask first, then write.
401
- lines.push(
402
- `After both answers exist (asked now or recovered from the ledger), persist them in ONE call: plans record-checkpoint (checkpoint: { transition: "implementation-review-configured", terminationCondition: "<answer>", reviewerCount: <integer> }).`,
403
- );
404
- } else {
405
- lines.push(`Completed rounds in this worktree: ${review?.completedRounds ?? 0} (hard cap 5).`);
406
- }
407
- const currentRound = review?.currentRoundId
408
- ? cp.reviewRounds.find((round) => round.roundId === review.currentRoundId)
409
- : undefined;
410
- if (currentRound) {
411
- const done = currentRound.lanes.filter((lane) => lane.status === "complete").map((lane) => lane.laneId);
412
- const pending = currentRound.lanes.filter((lane) => lane.status !== "complete").map((lane) => lane.laneId);
413
- lines.push(
414
- `Round ${currentRound.roundId} is in flight — complete lanes: ${done.join(", ") || "none"}; pending/failed lanes: ${pending.join(", ") || "none"}. Resume it with refine (role: "reviewer", target: "implementation", resumeRoundId: "${currentRound.roundId}") so completed lanes are reused, never re-run.`,
415
- );
416
- } else {
417
- lines.push(
418
- `Start the next round with refine (role: "reviewer", target: "implementation", reviewers: ${review?.reviewerCount ?? effectiveCount ?? "<configured count>"}) — do NOT pass a resumeRoundId unless resuming an interrupted round; an omitted reviewers argument falls back to the run's configured reviewerCount.`,
419
- );
330
+ // v0.6.1 (D-018/D-020): the implementation-review loop is gone; a
331
+ // legacy 0.6.0 checkpoint in this phase maps to execution completed.
332
+ // The run flips to done so the status line and run registry agree.
333
+ try {
334
+ setRunStatus(ctx.cwd, runId, "done");
335
+ } catch {
336
+ /* best-effort */
420
337
  }
421
- lines.push(
422
- `Record boundaries with plans record-checkpoint: review-consolidated → implementation-round-finished per round; completed (with evidence) when the termination condition is met.`,
338
+ ctx.ui.notify(
339
+ `${runId} finished under the removed v0.6.0 implementation-review loop; mapped to done. Its historical acceptance stands; no review loop to resume.`,
340
+ "info",
423
341
  );
424
- return { phaseLabel: "implementation-review", text: lines.join("\n") };
342
+ return null;
425
343
  }
426
344
 
427
345
  // planning / reviewing: rebuild the workflow context (R-002/R-003).
@@ -512,19 +430,7 @@ async function buildLegacyBrief(
512
430
  `This run predates durable checkpoints: treat every VC as unverified and re-run the execution handoff (execute_plan or /plans-execute) for explicit approval before writing any code.`,
513
431
  );
514
432
  }
515
- if (candidate.run.status === "done") {
516
- const ok = await ctx.ui.confirm(
517
- "Finished run with review artifacts",
518
- `${candidate.runId} is marked done and has review records, but completion of the implementation-review loop cannot be proven for legacy runs. Resume the review loop anyway?`,
519
- );
520
- if (!ok) {
521
- ctx.ui.notify("Cancelled; nothing was changed.", "info");
522
- return null;
523
- }
524
- lines.push(
525
- `Resume the implementation-review loop: ask the termination condition (questionId: "termination-condition") if unknown, then run rounds with refine (target: "implementation").`,
526
- );
527
- }
528
433
  lines.push(`Continue only the missing work; never restart the interview from scratch.`);
529
434
  return { phaseLabel: candidate.phaseLabel, text: lines.join("\n") };
530
435
  }
436
+