pi-plans 0.5.7 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +126 -0
- package/README.md +49 -39
- package/agents/ref-analyst.md +7 -4
- package/agents/reviewer.md +12 -3
- package/index.ts +74 -40
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +45 -58
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +63 -47
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +22 -10
- package/skills/debug-and-plan/SKILL.md +4 -4
- package/skills/plan-big/SKILL.md +5 -5
- package/skills/plan-normal/SKILL.md +5 -5
- package/skills/plan-small/SKILL.md +5 -5
- package/skills/plan-with-refs/SKILL.md +8 -8
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +154 -76
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +709 -705
- package/src/global-state.ts +304 -0
- package/src/guard.ts +16 -3
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +14 -72
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +40 -130
- package/src/resume.ts +15 -17
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +5 -4
- package/src/run-picker.ts +98 -0
- package/src/state.ts +380 -77
- package/src/subagent.ts +32 -1
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +78 -57
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-pros-cons.test.ts +147 -0
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +184 -0
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +43 -88
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +19 -49
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +63 -33
- package/tools/graph-aware-file-tools.ts +6 -4
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/src/refine-prompts.ts
CHANGED
|
@@ -17,11 +17,11 @@ export const REVIEWER_LENSES: readonly ReviewerLane[] = [
|
|
|
17
17
|
{ id: "verification", lens: "verification rigor, risks, and evidence gaps" },
|
|
18
18
|
] as const;
|
|
19
19
|
|
|
20
|
-
function buildSharedHeader(
|
|
21
|
-
const lensLine =
|
|
20
|
+
function buildSharedHeader(opts: RefinePromptInput): string {
|
|
21
|
+
const lensLine = opts.lens ? `\nReview lens: ${opts.lens}.` : "";
|
|
22
22
|
const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
|
|
23
23
|
const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
|
|
24
|
-
return `Goal:
|
|
24
|
+
return `Goal: review the plan against the repository and surface what needs the user's judgment.
|
|
25
25
|
|
|
26
26
|
Target: ${opts.planPath}
|
|
27
27
|
|
|
@@ -30,23 +30,6 @@ Authority boundary: read-only analysis only. Do not edit, write, delete, commit,
|
|
|
30
30
|
Evidence: inspect the repository with read, grep, find, and ls before judging the plan.${lensLine}${focusLine}${contextLine}`;
|
|
31
31
|
}
|
|
32
32
|
|
|
33
|
-
/** Shared header for the post-execution implementation review: the
|
|
34
|
-
* accepted plan is the contract, the IMPLEMENTATION in the worktree is
|
|
35
|
-
* under review. Findings must anchor to the plan's goals/acceptance
|
|
36
|
-
* criteria and explicitly assess delivery maturity. */
|
|
37
|
-
function buildImplementationSharedHeader(role: "reviewer" | "criticizer", opts: RefinePromptInput): string {
|
|
38
|
-
const lensLine = role === "reviewer" && opts.lens ? `\nReview lens: ${opts.lens}.` : "";
|
|
39
|
-
const focusLine = opts.focus ? `\n\nSpecific concerns from the main agent: ${opts.focus}` : "";
|
|
40
|
-
const contextLine = opts.context ? `\n\nContext: ${opts.context}` : "";
|
|
41
|
-
return `Goal: ${role === "reviewer" ? "review the implemented result in the worktree against the plan" : "stress-test the implemented result's assumptions"}.
|
|
42
|
-
|
|
43
|
-
Target: ${opts.planPath} (the accepted plan; the IMPLEMENTATION in the worktree is under review)
|
|
44
|
-
|
|
45
|
-
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
|
|
46
|
-
|
|
47
|
-
Evidence: inspect the repository with read, grep, find, and ls before judging the implementation. Judge the implementation against the plan's goals, verifier checklist, and acceptance criteria. Explicitly assess delivery maturity: did the executor ship a minimal MVP only, or refine for long-term growth (no stopgaps, long-term architectural decisions, missing tests, technical debt, production readiness)? Calibrate severity accordingly. Out-of-scope improvement ideas are low severity by default and must not be forced into findings.${lensLine}${focusLine}${contextLine}`;
|
|
48
|
-
}
|
|
49
|
-
|
|
50
33
|
export function reviewerLanes(count: number): ReviewerLane[] {
|
|
51
34
|
if (count === 3) return [...REVIEWER_LENSES];
|
|
52
35
|
if (count === 2) return [...REVIEWER_LENSES.slice(0, 2)];
|
|
@@ -54,47 +37,22 @@ export function reviewerLanes(count: number): ReviewerLane[] {
|
|
|
54
37
|
}
|
|
55
38
|
|
|
56
39
|
export function buildReviewerTask(opts: RefinePromptInput): string {
|
|
57
|
-
return `${buildSharedHeader(
|
|
58
|
-
|
|
59
|
-
Success criteria: return evidence-backed findings or explicitly say the plan holds up.
|
|
60
|
-
|
|
61
|
-
Output: Markdown, highest severity first. For each finding use this shape:
|
|
62
|
-
- \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
|
|
63
|
-
|
|
64
|
-
Surface at most five high-priority findings; list lower-severity findings after them. If the plan holds up, say so explicitly and list what you checked.
|
|
65
|
-
|
|
66
|
-
Plan file: ${opts.planPath}
|
|
67
|
-
|
|
68
|
-
---8<--- PLAN CONTENT ---8<---
|
|
69
|
-
${opts.planText}
|
|
70
|
-
---8<--- END PLAN CONTENT ---8<---`;
|
|
71
|
-
}
|
|
40
|
+
return `${buildSharedHeader(opts)}
|
|
72
41
|
|
|
73
|
-
|
|
74
|
-
return `${buildSharedHeader("criticizer", opts)}
|
|
42
|
+
Success criteria: return evidence-backed findings, plus the questions only the user can settle. If the plan holds up, say so explicitly and list what you checked.
|
|
75
43
|
|
|
76
|
-
|
|
44
|
+
Output: Markdown with exactly two top-level parts, in this order.
|
|
77
45
|
|
|
78
|
-
|
|
79
|
-
1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
|
|
80
|
-
2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the plan genuinely holds.
|
|
46
|
+
## Findings
|
|
81
47
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
---8<--- PLAN CONTENT ---8<---
|
|
85
|
-
${opts.planText}
|
|
86
|
-
---8<--- END PLAN CONTENT ---8<---`;
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
export function buildImplementationReviewerTask(opts: RefinePromptInput): string {
|
|
90
|
-
return `${buildImplementationSharedHeader("reviewer", opts)}
|
|
48
|
+
Highest severity first. For each finding use this shape:
|
|
49
|
+
- \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003); evidence: repo path/command or external source that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
|
|
91
50
|
|
|
92
|
-
|
|
51
|
+
Surface at most five high-priority findings; list lower-severity findings after them. Write "None." when there are none.
|
|
93
52
|
|
|
94
|
-
|
|
95
|
-
- \`F-###\` — severity: high | medium | low; affected plan IDs (e.g. R-001, I-003) or files; evidence: repo path/command that proves it; impact; recommended fix; suggested disposition (accept | reject | needs-discussion).
|
|
53
|
+
## Questions
|
|
96
54
|
|
|
97
|
-
|
|
55
|
+
At most five numbered questions (\`Q-1\`, \`Q-2\`, …) covering everything that needs the user's decision before the plan can be safely revised — hidden trade-offs, undetermined semantics, accept/reject calls on findings marked needs-discussion. Each question gets one line of why it matters, phrased so a user with repo access can answer it concretely. Never rhetorical; never questions the repository itself already answers. Stop earlier if nothing genuinely needs the user.
|
|
98
56
|
|
|
99
57
|
Plan file: ${opts.planPath}
|
|
100
58
|
|
|
@@ -144,7 +102,7 @@ Local path (your working directory): ${opts.localPath}
|
|
|
144
102
|
|
|
145
103
|
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents. Stay inside the reference directory.
|
|
146
104
|
|
|
147
|
-
Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite
|
|
105
|
+
Evidence: inspect the reference with read, grep, find, and ls before judging it. Cite evidence for every claim in the medium's format — code: <relative-path>:<line>; papers: section/theorem/table numbers with a short quote; blogs/docs: the heading or quoted passage. Quote only what you verified. Medium-aware deep-read: repos go through entry points, core modules, tests, and configuration; papers through claims, method, limitations, and experiments; blogs/docs through technique, measurements, and caveats. Theoretical grounding counts — an algorithm, a formal property, or a measured tradeoff is as adoptable as an implementation pattern.${contextLine}${languageLine}
|
|
148
106
|
|
|
149
107
|
Success criteria: a structured analysis the main agent can paste into REF_ANALYSIS.md and turn into adoption questions.
|
|
150
108
|
|
|
@@ -157,23 +115,7 @@ Section contracts:
|
|
|
157
115
|
- Key Mechanisms And Design Tradeoffs: the mechanisms that make it work and the tradeoffs they embody.
|
|
158
116
|
- Adoptable Ideas For The Target Repo: concrete, portable ideas ranked by expected value; name the target-repo surface each would touch.
|
|
159
117
|
- Pitfalls And Anti-Patterns: what to avoid when borrowing; failure modes the reference itself documents or exhibits.
|
|
160
|
-
- Evidence Citations: the
|
|
118
|
+
- Evidence Citations: the evidence references backing the claims above (file:line for code; section/theorem/table + quote for papers; heading/quote for blogs and docs).
|
|
161
119
|
- Coverage: which parts of the reference you actually read versus skipped.
|
|
162
120
|
- Evidence Gaps: what you could not determine from the reference alone.`;
|
|
163
121
|
}
|
|
164
|
-
|
|
165
|
-
export function buildImplementationCriticizerTask(opts: RefinePromptInput): string {
|
|
166
|
-
return `${buildImplementationSharedHeader("criticizer", opts)}
|
|
167
|
-
|
|
168
|
-
Success criteria: return concrete, answerable questions only; never rewrite the plan or the implementation.
|
|
169
|
-
|
|
170
|
-
Output: Markdown in exactly this shape:
|
|
171
|
-
1. A summary of your core criticism in at most three sentences, highlighting the single most important point.
|
|
172
|
-
2. Then at most five adaptive questions, numbered, each with one line of why it matters. Questions must be answerable by a user with repo access — never rhetorical. Stop earlier if the implementation genuinely holds.
|
|
173
|
-
|
|
174
|
-
Plan file: ${opts.planPath}
|
|
175
|
-
|
|
176
|
-
---8<--- PLAN CONTENT ---8<---
|
|
177
|
-
${opts.planText}
|
|
178
|
-
---8<--- END PLAN CONTENT ---8<---`;
|
|
179
|
-
}
|
package/src/refine-ui-helpers.ts
CHANGED
|
@@ -10,11 +10,30 @@
|
|
|
10
10
|
const ELLIPSIS = "…";
|
|
11
11
|
|
|
12
12
|
/**
|
|
13
|
-
* ECMA-48
|
|
14
|
-
*
|
|
15
|
-
*
|
|
13
|
+
* ECMA-48 escape sequences. ALL of them render at zero width, so styling and
|
|
14
|
+
* control payloads must never leak into width math nor be split mid-sequence.
|
|
15
|
+
*
|
|
16
|
+
* Three families are recognised:
|
|
17
|
+
* - CSI: `ESC [ params intermediates final` (SGR colours, cursor moves)
|
|
18
|
+
* - String-terminated: `ESC ] _ P X ^` … `BEL`|`ST` (OSC, APC, DCS, SOS, PM)
|
|
19
|
+
* - Simple: `ESC` intermediates `final` (`ESC c`, `ESC (B`, `ESC 7`)
|
|
20
|
+
*
|
|
21
|
+
* The string family matters beyond colour: pi's `CURSOR_MARKER` is an APC
|
|
22
|
+
* sequence (`ESC _ pi:c BEL`). Counting its payload as visible text made every
|
|
23
|
+
* row carrying the marker — e.g. a focused search Input — measure several
|
|
24
|
+
* columns too wide, which pushed that row's right-hand border out of
|
|
25
|
+
* alignment. The terminator is required: a well-formed sequence always has
|
|
26
|
+
* one, and refusing malformed ones keeps the fallback (the simple family)
|
|
27
|
+
* from swallowing real text.
|
|
16
28
|
*/
|
|
17
|
-
const
|
|
29
|
+
const ESCAPE_PATTERN = new RegExp(
|
|
30
|
+
[
|
|
31
|
+
"\\x1b\\[[\\x30-\\x3f]*[\\x20-\\x2f]*[\\x40-\\x7e]", // CSI
|
|
32
|
+
"\\x1b[\\]PX^_][^\\x07\\x1b]*(?:\\x07|\\x1b\\\\)", // OSC / APC / DCS / SOS / PM
|
|
33
|
+
"\\x1b[\\x20-\\x2f]*[\\x30-\\x7e]", // simple escapes
|
|
34
|
+
].join("|"),
|
|
35
|
+
"g",
|
|
36
|
+
);
|
|
18
37
|
|
|
19
38
|
interface AnsiPart {
|
|
20
39
|
kind: "csi" | "text";
|
|
@@ -24,7 +43,7 @@ interface AnsiPart {
|
|
|
24
43
|
function splitAnsi(text: string): AnsiPart[] {
|
|
25
44
|
const parts: AnsiPart[] = [];
|
|
26
45
|
let last = 0;
|
|
27
|
-
for (const match of text.matchAll(
|
|
46
|
+
for (const match of text.matchAll(ESCAPE_PATTERN)) {
|
|
28
47
|
const start = match.index ?? 0;
|
|
29
48
|
if (start > last) parts.push({ kind: "text", value: text.slice(last, start) });
|
|
30
49
|
parts.push({ kind: "csi", value: match[0] });
|
package/src/refine-ui-state.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
|
|
2
2
|
|
|
3
|
-
export type RefineOverlayRole = "reviewer" | "
|
|
3
|
+
export type RefineOverlayRole = "reviewer" | "refs";
|
|
4
4
|
export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
|
|
5
5
|
export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
|
|
6
6
|
|
package/src/refine-ui.ts
CHANGED
|
@@ -166,7 +166,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
|
|
|
166
166
|
const complete = lanes.filter((lane) => lane.status === "complete").length;
|
|
167
167
|
const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
|
|
168
168
|
const running = lanes.filter((lane) => lane.status === "running").length;
|
|
169
|
-
const title = role === "reviewer" ? "Reviewer" :
|
|
169
|
+
const title = role === "reviewer" ? "Reviewer" : "Refs";
|
|
170
170
|
const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
|
|
171
171
|
const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
|
|
172
172
|
return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
|
package/src/resume-command.ts
CHANGED
|
@@ -12,14 +12,14 @@
|
|
|
12
12
|
* - R-007: after every dialog the world is re-checked; one kickoff max.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
-
import type {
|
|
15
|
+
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
16
16
|
import * as fs from "node:fs";
|
|
17
17
|
import { existsSync } from "node:fs";
|
|
18
18
|
import * as path from "node:path";
|
|
19
19
|
import { loadExecutionFromCheckpoint } from "./exec.ts";
|
|
20
|
-
import { bindRun } from "./run-context.ts";
|
|
20
|
+
import { bindRun, boundRunId } from "./run-context.ts";
|
|
21
21
|
import { acquireOwnership, OwnershipError, releaseOwnership } from "./run-ownership.ts";
|
|
22
|
-
import { loadConfig, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
|
|
22
|
+
import { loadConfig, resolveArtifactRoot, resolveStateRootOrNull, setRunStatus, updateRunWorkdir } from "./state.ts";
|
|
23
23
|
import {
|
|
24
24
|
applyMigration,
|
|
25
25
|
mutateCheckpoint,
|
|
@@ -35,12 +35,12 @@ import {
|
|
|
35
35
|
reconcileCheckpointWithLedger,
|
|
36
36
|
type ResumeCandidate,
|
|
37
37
|
} from "./resume.ts";
|
|
38
|
+
import { messaging } from "./messaging.ts";
|
|
38
39
|
|
|
39
40
|
/** One kickoff per invocation; repeat invocations are blocked by the idle check. */
|
|
40
41
|
let inFlight = false;
|
|
41
42
|
|
|
42
43
|
export async function resumePlansCommand(
|
|
43
|
-
pi: ExtensionAPI,
|
|
44
44
|
ctx: ExtensionContext,
|
|
45
45
|
baseDir: string,
|
|
46
46
|
): Promise<void> {
|
|
@@ -50,13 +50,13 @@ export async function resumePlansCommand(
|
|
|
50
50
|
}
|
|
51
51
|
inFlight = true;
|
|
52
52
|
try {
|
|
53
|
-
await run(
|
|
53
|
+
await run(ctx, baseDir);
|
|
54
54
|
} finally {
|
|
55
55
|
inFlight = false;
|
|
56
56
|
}
|
|
57
57
|
}
|
|
58
58
|
|
|
59
|
-
async function run(
|
|
59
|
+
async function run(ctx: ExtensionContext, baseDir: string): Promise<void> {
|
|
60
60
|
if (!ctx.hasUI) {
|
|
61
61
|
ctx.ui.notify?.("/resume-plans needs an interactive session (TUI/RPC); print/json cannot resume.", "warning");
|
|
62
62
|
return;
|
|
@@ -73,7 +73,11 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
|
|
|
73
73
|
return;
|
|
74
74
|
}
|
|
75
75
|
|
|
76
|
-
|
|
76
|
+
// v0.6.0 (D-1): a session-bound resumable run resumes directly; the
|
|
77
|
+
// active-pointer auto-win is gone — ambiguity opens the descriptive form.
|
|
78
|
+
const bound = boundRunId(ctx.sessionManager, ctx.cwd);
|
|
79
|
+
const boundCandidate = bound ? candidates.find((candidate) => candidate.runId === bound) ?? null : null;
|
|
80
|
+
let candidate = boundCandidate ?? pickDefaultCandidate(ctx.cwd, candidates);
|
|
77
81
|
if (candidate === null) {
|
|
78
82
|
const labels = candidates.map((entry, index) => {
|
|
79
83
|
const cross = entry.crossWorktree ? " · cross-worktree" : "";
|
|
@@ -101,7 +105,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
|
|
|
101
105
|
}
|
|
102
106
|
if (candidate.checkpointStatus === "corrupt") {
|
|
103
107
|
ctx.ui.notify(
|
|
104
|
-
`Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/
|
|
108
|
+
`Checkpoint for ${candidate.runId} is corrupt (${candidate.checkpointError ?? "unknown"}). Repair or remove .git/pi-plans/runs/${candidate.runId}/checkpoint.json explicitly; nothing was changed.`,
|
|
105
109
|
"error",
|
|
106
110
|
);
|
|
107
111
|
return;
|
|
@@ -155,7 +159,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
|
|
|
155
159
|
}
|
|
156
160
|
}
|
|
157
161
|
|
|
158
|
-
const brief = await buildBrief(
|
|
162
|
+
const brief = await buildBrief(ctx, baseDir, candidate);
|
|
159
163
|
if (brief === null) {
|
|
160
164
|
releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
|
|
161
165
|
return; // buildBrief reported the specific problem
|
|
@@ -163,7 +167,7 @@ async function run(pi: ExtensionAPI, ctx: ExtensionContext, baseDir: string): Pr
|
|
|
163
167
|
|
|
164
168
|
// Exactly one kickoff (R-007). The idle re-check happens above and in
|
|
165
169
|
// buildBrief's dialogs; the message itself starts the continuation.
|
|
166
|
-
await
|
|
170
|
+
await messaging().sendUserMessage(brief.text);
|
|
167
171
|
ctx.ui.notify(`Resumed ${candidate.runId} (${brief.phaseLabel}).`, "info");
|
|
168
172
|
} catch (error) {
|
|
169
173
|
releaseOwnership(ctx.cwd, candidate.runId, ownerToken);
|
|
@@ -185,13 +189,12 @@ export function migrateRunIntoCurrentWorktree(workdir: string, candidate: Resume
|
|
|
185
189
|
const stateRoot = loadConfigShared(workdir);
|
|
186
190
|
if (stateRoot === null) return null;
|
|
187
191
|
const config = loadConfig(stateRoot);
|
|
188
|
-
|
|
189
|
-
if (!path.isAbsolute(artifactRoot)) artifactRoot = path.resolve(workdir, artifactRoot);
|
|
192
|
+
const artifactRoot = resolveArtifactRoot(workdir, config.artifact_root);
|
|
190
193
|
const sourceDir = candidate.run.artifact_dir;
|
|
191
194
|
if (!existsSync(sourceDir)) return null;
|
|
192
195
|
// Artifacts already in a shared location stay put.
|
|
193
196
|
const gitRoot = findGitCommonDir(workdir);
|
|
194
|
-
if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "
|
|
197
|
+
if (gitRoot !== null && isInside(sourceDir, path.join(gitRoot, "pi-plans"))) return sourceDir;
|
|
195
198
|
const targetDir = path.join(artifactRoot, path.basename(sourceDir));
|
|
196
199
|
for (const entry of fs.readdirSync(sourceDir, { withFileTypes: true })) {
|
|
197
200
|
if (!entry.isFile()) continue;
|
|
@@ -257,46 +260,7 @@ export interface ResumeBrief {
|
|
|
257
260
|
text: string;
|
|
258
261
|
}
|
|
259
262
|
|
|
260
|
-
/** D-5 crash-window recovery: rebuild the implementation-review loop
|
|
261
|
-
* configuration from the checkpoint plus the decisions ledger. Answers land
|
|
262
|
-
* in the ledger first (stable questionIds), the combined record-checkpoint
|
|
263
|
-
* second; between the two, this resolver reconstructs what is known and
|
|
264
|
-
* reports which questions are still missing. Exported for tests. */
|
|
265
|
-
export function resolveImplReviewConfig(
|
|
266
|
-
review: { terminationCondition?: string; reviewerCount?: number; completedRounds: number } | undefined,
|
|
267
|
-
ledger: { questionId?: string; answer?: string }[],
|
|
268
|
-
): {
|
|
269
|
-
condition: string | undefined;
|
|
270
|
-
conditionFromLedger: boolean;
|
|
271
|
-
reviewerCount: number | undefined;
|
|
272
|
-
reviewerCountFromLedger: boolean;
|
|
273
|
-
missing: Array<"termination-condition" | "impl-review-reviewer-count">;
|
|
274
|
-
} {
|
|
275
|
-
const latest = (id: string): string | undefined =>
|
|
276
|
-
[...ledger].reverse().find((entry) => entry.questionId === id)?.answer;
|
|
277
|
-
const ledgerCondition = latest("termination-condition");
|
|
278
|
-
const ledgerCountRaw = latest("impl-review-reviewer-count");
|
|
279
|
-
const ledgerCount = ledgerCountRaw === undefined ? undefined : Number.parseInt(ledgerCountRaw, 10);
|
|
280
|
-
const ledgerCountValid = ledgerCount !== undefined && Number.isInteger(ledgerCount) && ledgerCount >= 1 && ledgerCount <= 3;
|
|
281
|
-
const condition = review?.terminationCondition ?? ledgerCondition;
|
|
282
|
-
const reviewerCount = review?.reviewerCount ?? (ledgerCountValid ? ledgerCount : undefined);
|
|
283
|
-
// Both questions are per-run configuration: a count is missing whenever it
|
|
284
|
-
// is unknown (including legacy checkpoints persisted before 0.5.4),
|
|
285
|
-
// independent of where the condition came from.
|
|
286
|
-
const missing: Array<"termination-condition" | "impl-review-reviewer-count"> = [];
|
|
287
|
-
if (condition === undefined) missing.push("termination-condition");
|
|
288
|
-
if (reviewerCount === undefined) missing.push("impl-review-reviewer-count");
|
|
289
|
-
return {
|
|
290
|
-
condition,
|
|
291
|
-
conditionFromLedger: review?.terminationCondition === undefined && ledgerCondition !== undefined,
|
|
292
|
-
reviewerCount,
|
|
293
|
-
reviewerCountFromLedger: review?.reviewerCount === undefined && ledgerCountValid,
|
|
294
|
-
missing,
|
|
295
|
-
};
|
|
296
|
-
}
|
|
297
|
-
|
|
298
263
|
async function buildBrief(
|
|
299
|
-
pi: ExtensionAPI,
|
|
300
264
|
ctx: ExtensionContext,
|
|
301
265
|
baseDir: string,
|
|
302
266
|
candidate: ResumeCandidate,
|
|
@@ -323,7 +287,7 @@ async function buildBrief(
|
|
|
323
287
|
bindRun(ctx.sessionManager, ctx.cwd, runId);
|
|
324
288
|
|
|
325
289
|
if (cp.phase === "executing") {
|
|
326
|
-
const load = loadExecutionFromCheckpoint(
|
|
290
|
+
const load = loadExecutionFromCheckpoint(ctx, runId);
|
|
327
291
|
if (load.status === "loaded") {
|
|
328
292
|
if (run.status === "stopped" || run.status === "accepted") {
|
|
329
293
|
try {
|
|
@@ -334,14 +298,22 @@ async function buildBrief(
|
|
|
334
298
|
}
|
|
335
299
|
const doneList = (load.doneVcIds ?? []).join(", ") || "none";
|
|
336
300
|
const reverify = load.reverifyAll
|
|
337
|
-
? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously
|
|
301
|
+
? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
|
|
338
302
|
: `\nPreviously verified and still valid: ${doneList}.`;
|
|
339
|
-
const paused = load.pausedReason ? `\nExecution
|
|
303
|
+
const paused = load.pausedReason ? `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.` : "";
|
|
304
|
+
const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
|
|
340
305
|
return {
|
|
341
306
|
phaseLabel: "executing",
|
|
342
|
-
text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}\nFollow the execution-loop contract:
|
|
307
|
+
text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\nFollow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the completion auditor verify the checks. The current wave and remaining tasks are injected each turn.`,
|
|
343
308
|
};
|
|
344
309
|
}
|
|
310
|
+
if (load.legacyDelegate) {
|
|
311
|
+
ctx.ui.notify(
|
|
312
|
+
`Cannot resume ${runId} directly: it was mid-flight under a v0.6.0 delegated executor (removed in v0.6.1). Run /plans-execute to re-approve the handoff; execution restarts from the first task (0.6.0 progress cannot map onto the task tree).`,
|
|
313
|
+
"warning",
|
|
314
|
+
);
|
|
315
|
+
return null;
|
|
316
|
+
}
|
|
345
317
|
if (load.status === "plan-missing" || load.status === "plan-mismatch") {
|
|
346
318
|
ctx.ui.notify(`Cannot resume ${runId}: ${load.error ?? "plan file missing"}.`, "error");
|
|
347
319
|
return null;
|
|
@@ -355,69 +327,19 @@ async function buildBrief(
|
|
|
355
327
|
}
|
|
356
328
|
|
|
357
329
|
if (cp.phase === "implementation-review") {
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
// ledger under stable questionIds. Rebuild from the ledger, ask ONLY the
|
|
366
|
-
// missing question(s), then persist BOTH in one combined write. The
|
|
367
|
-
// re-ask guard on the checkpoint (termination condition already
|
|
368
|
-
// configured) rejects duplicate writes, so the combined write happens
|
|
369
|
-
// only while the field is genuinely absent.
|
|
370
|
-
const ledger = readDecisionLedger(ctx.cwd, runId);
|
|
371
|
-
const resolved = resolveImplReviewConfig(review, ledger);
|
|
372
|
-
const effectiveCondition = resolved.condition;
|
|
373
|
-
const effectiveCount = resolved.reviewerCount;
|
|
374
|
-
if (effectiveCondition === undefined) {
|
|
375
|
-
lines.push(
|
|
376
|
-
`The termination condition was never chosen. Ask it now via ask_choice (autoComplete: false, questionId: "termination-condition"): "How should the implementation-review loop terminate?" Options: goal wait (recommended) / until no high-severity finding (hard cap 5 rounds) / 1 / 2 / 3 rounds.`,
|
|
377
|
-
);
|
|
378
|
-
} else {
|
|
379
|
-
lines.push(`Termination condition: ${effectiveCondition}${resolved.conditionFromLedger ? " (recovered from the decisions ledger)" : ""}`);
|
|
380
|
-
}
|
|
381
|
-
if (review?.reviewerCount === undefined) {
|
|
382
|
-
if (effectiveCount !== undefined && Number.isInteger(effectiveCount) && effectiveCount >= 1 && effectiveCount <= 3) {
|
|
383
|
-
lines.push(`Reviewer count: ${effectiveCount} (recovered from the decisions ledger; not yet persisted).`);
|
|
384
|
-
} else {
|
|
385
|
-
lines.push(
|
|
386
|
-
`The reviewer count was never chosen. Ask it via ask_choice (autoComplete: false, allowOther: false, questionId: "impl-review-reviewer-count", digit labels "1"/"2"/"3"): "How many concurrent reviewers should each implementation-review round use?" — recommended default follows this run's skill: plan-big / plan-with-refs → 3, others → 1.`,
|
|
387
|
-
);
|
|
388
|
-
}
|
|
389
|
-
} else {
|
|
390
|
-
lines.push(`Reviewers per round: ${review.reviewerCount}`);
|
|
391
|
-
}
|
|
392
|
-
if (condition === undefined) {
|
|
393
|
-
// Only when the checkpoint is still unconfigured: the combined write
|
|
394
|
-
// (both answers) is one transaction. When only terminationCondition is
|
|
395
|
-
// missing but a ledger answer exists for both, write both; when a
|
|
396
|
-
// question is genuinely unanswered, ask first, then write.
|
|
397
|
-
lines.push(
|
|
398
|
-
`After both answers exist (asked now or recovered from the ledger), persist them in ONE call: plans record-checkpoint (checkpoint: { transition: "implementation-review-configured", terminationCondition: "<answer>", reviewerCount: <integer> }).`,
|
|
399
|
-
);
|
|
400
|
-
} else {
|
|
401
|
-
lines.push(`Completed rounds in this worktree: ${review?.completedRounds ?? 0} (hard cap 5).`);
|
|
402
|
-
}
|
|
403
|
-
const currentRound = review?.currentRoundId
|
|
404
|
-
? cp.reviewRounds.find((round) => round.roundId === review.currentRoundId)
|
|
405
|
-
: undefined;
|
|
406
|
-
if (currentRound) {
|
|
407
|
-
const done = currentRound.lanes.filter((lane) => lane.status === "complete").map((lane) => lane.laneId);
|
|
408
|
-
const pending = currentRound.lanes.filter((lane) => lane.status !== "complete").map((lane) => lane.laneId);
|
|
409
|
-
lines.push(
|
|
410
|
-
`Round ${currentRound.roundId} is in flight — complete lanes: ${done.join(", ") || "none"}; pending/failed lanes: ${pending.join(", ") || "none"}. Resume it with refine (role: "reviewer", target: "implementation", resumeRoundId: "${currentRound.roundId}") so completed lanes are reused, never re-run.`,
|
|
411
|
-
);
|
|
412
|
-
} else {
|
|
413
|
-
lines.push(
|
|
414
|
-
`Start the next round with refine (role: "reviewer", target: "implementation", reviewers: ${review?.reviewerCount ?? effectiveCount ?? "<configured count>"}) — do NOT pass a resumeRoundId unless resuming an interrupted round; an omitted reviewers argument falls back to the run's configured reviewerCount.`,
|
|
415
|
-
);
|
|
330
|
+
// v0.6.1 (D-018/D-020): the implementation-review loop is gone; a
|
|
331
|
+
// legacy 0.6.0 checkpoint in this phase maps to execution completed.
|
|
332
|
+
// The run flips to done so the status line and run registry agree.
|
|
333
|
+
try {
|
|
334
|
+
setRunStatus(ctx.cwd, runId, "done");
|
|
335
|
+
} catch {
|
|
336
|
+
/* best-effort */
|
|
416
337
|
}
|
|
417
|
-
|
|
418
|
-
|
|
338
|
+
ctx.ui.notify(
|
|
339
|
+
`${runId} finished under the removed v0.6.0 implementation-review loop; mapped to done. Its historical acceptance stands; no review loop to resume.`,
|
|
340
|
+
"info",
|
|
419
341
|
);
|
|
420
|
-
return
|
|
342
|
+
return null;
|
|
421
343
|
}
|
|
422
344
|
|
|
423
345
|
// planning / reviewing: rebuild the workflow context (R-002/R-003).
|
|
@@ -508,19 +430,7 @@ async function buildLegacyBrief(
|
|
|
508
430
|
`This run predates durable checkpoints: treat every VC as unverified and re-run the execution handoff (execute_plan or /plans-execute) for explicit approval before writing any code.`,
|
|
509
431
|
);
|
|
510
432
|
}
|
|
511
|
-
if (candidate.run.status === "done") {
|
|
512
|
-
const ok = await ctx.ui.confirm(
|
|
513
|
-
"Finished run with review artifacts",
|
|
514
|
-
`${candidate.runId} is marked done and has review records, but completion of the implementation-review loop cannot be proven for legacy runs. Resume the review loop anyway?`,
|
|
515
|
-
);
|
|
516
|
-
if (!ok) {
|
|
517
|
-
ctx.ui.notify("Cancelled; nothing was changed.", "info");
|
|
518
|
-
return null;
|
|
519
|
-
}
|
|
520
|
-
lines.push(
|
|
521
|
-
`Resume the implementation-review loop: ask the termination condition (questionId: "termination-condition") if unknown, then run rounds with refine (target: "implementation").`,
|
|
522
|
-
);
|
|
523
|
-
}
|
|
524
433
|
lines.push(`Continue only the missing work; never restart the interview from scratch.`);
|
|
525
434
|
return { phaseLabel: candidate.phaseLabel, text: lines.join("\n") };
|
|
526
435
|
}
|
|
436
|
+
|
package/src/resume.ts
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
import * as fs from "node:fs";
|
|
10
10
|
import { existsSync, readFileSync } from "node:fs";
|
|
11
11
|
import * as path from "node:path";
|
|
12
|
-
import { getRun, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
|
|
12
|
+
import { getRun, listRuns, readActive, resolveStateRootOrNull, runDirPath, type RunInfo } from "./state.ts";
|
|
13
13
|
import { loadCheckpoint, mutateCheckpoint, type WorkflowCheckpoint } from "./workflow-state.ts";
|
|
14
14
|
|
|
15
15
|
export interface ResumeCandidate {
|
|
@@ -81,22 +81,17 @@ function planVersionOf(checkpoint: WorkflowCheckpoint | null, run: RunInfo): num
|
|
|
81
81
|
/**
|
|
82
82
|
* Enumerate resumable runs for the repo containing `workdir`. Corrupt
|
|
83
83
|
* checkpoints are surfaced (not hidden) so the command can report them;
|
|
84
|
-
* read errors never abort discovery of other runs.
|
|
84
|
+
* read errors never abort discovery of other runs. v0.6.0: enumeration is
|
|
85
|
+
* driven by the filesystem-derived registry (`listRuns`) so ordering and
|
|
86
|
+
* corrupt-run tolerance match every other multi-run surface.
|
|
85
87
|
*/
|
|
86
88
|
export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
87
89
|
const stateRoot = resolveStateRootOrNull(workdir);
|
|
88
90
|
if (stateRoot === null) return [];
|
|
89
|
-
const runsDir = path.join(stateRoot, "runs");
|
|
90
|
-
if (!existsSync(runsDir)) return [];
|
|
91
91
|
const active = readActive(workdir);
|
|
92
92
|
const candidates: ResumeCandidate[] = [];
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
entries = fs.readdirSync(runsDir);
|
|
96
|
-
} catch {
|
|
97
|
-
return [];
|
|
98
|
-
}
|
|
99
|
-
for (const runId of entries) {
|
|
93
|
+
for (const summary of listRuns(workdir)) {
|
|
94
|
+
const runId = summary.run_id;
|
|
100
95
|
const run = getRun(workdir, runId);
|
|
101
96
|
if (!run) continue;
|
|
102
97
|
const runDir = runDirPath(workdir, runId);
|
|
@@ -130,7 +125,7 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
|
130
125
|
updatedAt: checkpoint?.updatedAt ?? run.updated_at,
|
|
131
126
|
});
|
|
132
127
|
}
|
|
133
|
-
//
|
|
128
|
+
// Newest updated first; the active run (registry hint) gets priority.
|
|
134
129
|
candidates.sort((a, b) => {
|
|
135
130
|
const aActive = active?.run_id === a.runId ? 1 : 0;
|
|
136
131
|
const bActive = active?.run_id === b.runId ? 1 : 0;
|
|
@@ -140,14 +135,17 @@ export function listResumeCandidates(workdir: string): ResumeCandidate[] {
|
|
|
140
135
|
return candidates;
|
|
141
136
|
}
|
|
142
137
|
|
|
143
|
-
/**
|
|
138
|
+
/**
|
|
139
|
+
* Pick the default candidate (v0.6.0 D-1): the SESSION BINDING is resolved by
|
|
140
|
+
* the command (it owns the SessionManager); here a unique candidate goes
|
|
141
|
+
* direct and everything else is ambiguous — the active-pointer auto-win is
|
|
142
|
+
* gone (multi-run workdirs must not silently pick the registry hint).
|
|
143
|
+
*/
|
|
144
144
|
export function pickDefaultCandidate(workdir: string, candidates: ResumeCandidate[]): ResumeCandidate | null {
|
|
145
|
+
void workdir;
|
|
145
146
|
if (candidates.length === 0) return null;
|
|
146
|
-
const active = readActive(workdir);
|
|
147
|
-
const activeCandidate = active ? candidates.find((candidate) => candidate.runId === active.run_id) ?? null : null;
|
|
148
|
-
if (activeCandidate !== null) return activeCandidate;
|
|
149
147
|
if (candidates.length === 1) return candidates[0]!;
|
|
150
|
-
return null; // ambiguous: the command must ask
|
|
148
|
+
return null; // ambiguous: the command must ask (binding first, then form)
|
|
151
149
|
}
|
|
152
150
|
|
|
153
151
|
export interface DecisionLedgerEntry {
|