pi-plans 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/auditor.ts CHANGED
@@ -1,32 +1,58 @@
1
1
  /**
2
- * Completion auditor (v0.6.1): when every task reaches a terminal state, an
3
- * independent read-only subagent verifies the plan's verification checks
4
- * against the worktree. Failed checks roll their covered tasks back to
5
- * pending (exclusively inside the audit flow); three failed rounds pause the
6
- * run for the user — or terminate it as stopped under auto-approve/headless
7
- * so pipelines never hang.
2
+ * Execution reviewer (v0.8): when every task
3
+ * reaches a terminal state, an independent read-only subagent verifies the
4
+ * plan's verification checks against the worktree. Failed checks roll their
5
+ * covered tasks back to pending (exclusively inside the review flow). The
6
+ * loop is bounded at REVIEW_MAX_ROUNDS committed rounds; exhaustion pauses
7
+ * the run for the user in every mode (fail-closed, never a hang and never a
8
+ * silent stop) — only an explicit /plans-execute confirmation grants a fresh
9
+ * budget.
8
10
  */
9
11
 
10
12
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
11
13
  import * as fs from "node:fs";
12
- import { auditRollbackSet, auditableChecks, skippedPassCheckIds, type TaskView } from "./tasks.ts";
14
+ import * as path from "node:path";
15
+ import { auditableChecks, skippedPassCheckIds, type TaskView } from "./tasks.ts";
13
16
  import type { CheckItem } from "./plan.ts";
14
17
  import { messaging } from "./messaging.ts";
15
18
 
16
- export const AUDIT_MAX_ROUNDS = 3;
19
+ /** Budget cap: committed rounds per user-granted budget. Discarded (fingerprint-changed) attempts do not count. */
20
+ export const REVIEW_MAX_ROUNDS = 5;
21
+
22
+ /** Legacy alias for one release: checkpoints and old builds still know this name. */
23
+ export const AUDIT_MAX_ROUNDS = REVIEW_MAX_ROUNDS;
17
24
 
18
25
  export interface AuditOutcome {
19
26
  round: number;
20
27
  passed: string[];
21
28
  failed: string[];
22
- /** Rolled-back task ids (empty when the audit passed). */
23
- rolledBack: string[];
29
+ /** Checks whose report carried no readable verdict, or whose evidence was
30
+ * inconclusive. Never treated as `failed`: see src/exec.ts. */
31
+ undeterminable: string[];
24
32
  report: string;
25
33
  }
26
34
 
35
+ /** Parse the verdict lines of an audit report against the checks that are
36
+ * still pending. A pending check with no readable verdict is
37
+ * `undeterminable`, never `failed` — the caller must be able to tell "the
38
+ * work is wrong" apart from "I could not read the answer". */
39
+ export interface ParsedAudit {
40
+ passed: string[];
41
+ failed: string[];
42
+ undeterminable: string[];
43
+ }
44
+
45
+ /** Checks this round must judge: auditable (covering at least one task) and
46
+ * not already done. The brief and the coverage self-check both use this, so a
47
+ * well-formed round-2 report that omits an already-done check is not mistaken
48
+ * for a contract violation. */
49
+ export function auditablePendingChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
50
+ return auditableChecks(checklist, tasks).filter((item) => !item.done);
51
+ }
52
+
27
53
  /** Build the audit brief for the read-only subagent. Exported for tests. */
28
54
  export function buildAuditTask(planPath: string, checklist: CheckItem[], tasks: TaskView[], round: number): string {
29
- const checks = auditableChecks(checklist, tasks)
55
+ const checks = auditablePendingChecks(checklist, tasks)
30
56
  .map((check) => `- \`${check.id}\`: ${check.text}`)
31
57
  .join("\n");
32
58
  return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks.
@@ -37,51 +63,54 @@ Authority boundary: read-only analysis only. Do not edit, write, delete, commit,
37
63
 
38
64
  Evidence: inspect the repository with read, grep, find, ls, and targeted commands (bash is not granted — rely on the read tools) before judging each check. Tests may be referenced from their recorded evidence; do not re-run them.
39
65
 
40
- Checks to verify (only these; checks covering no task are excluded):
66
+ Checks to verify (only these; checks covering no task and checks already satisfied in an earlier round are excluded):
41
67
 
42
68
  ${checks}
43
69
 
44
- Output: Markdown with exactly one section per check, in checklist order:
70
+ Output: Markdown with exactly one section per check, in the order above:
45
71
 
46
- - \`VC-###\` — verdict: pass | fail; evidence: <repo path/command proving it>; note: <one line>.
72
+ - \`VC-###\` — verdict: pass | fail | undeterminable; evidence: <repo path/command proving it>; note: <one line>.
47
73
 
48
- Every check needs a verdict backed by evidence you actually inspected. If the evidence is inconclusive, verdict is fail with what is missing.`;
74
+ Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence.`;
49
75
  }
50
76
 
51
77
  /** Parse the audit subagent's verdict lines. Exported for tests. */
52
- export function parseAuditReport(report: string, checklist: CheckItem[]): { passed: string[]; failed: string[] } {
78
+ export function parseAuditReport(report: string, pendingVcIds: string[]): ParsedAudit {
53
79
  const passed: string[] = [];
54
80
  const failed: string[] = [];
55
- const known = new Set(checklist.map((item) => item.id));
56
- for (const match of report.matchAll(/`?(VC-\d+)`?[^\n]*?verdict:\s*(pass|fail)/gi)) {
81
+ const known = new Set(pendingVcIds.map((id) => id.toUpperCase()));
82
+ // The auditor writes Markdown, so a verdict may carry emphasis
83
+ // (`**pass**`, `*pass*`, `_pass_`, `` `pass` ``). Requiring a bare token
84
+ // silently discarded such verdicts, and the caller's fail-closed rule then
85
+ // marked every check failed and rolled the whole run back -- reporting
86
+ // correct work as failure. Tolerate the markers; \b keeps `passed` and
87
+ // `passing` from matching.
88
+ for (const match of report.matchAll(/`?(VC-\d+)`?[^\n]*?verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/gi)) {
57
89
  const id = match[1].toUpperCase();
58
90
  if (!known.has(id)) continue;
59
- (match[2].toLowerCase() === "pass" ? passed : failed).push(id);
91
+ const verdict = match[2].toLowerCase();
92
+ if (verdict === "pass") passed.push(id);
93
+ else if (verdict === "fail") failed.push(id);
60
94
  }
61
- // Dedupe, keep first occurrence order; a check with conflicting verdicts fails.
62
- for (const id of [...passed]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
63
- return { passed: [...new Set(passed)], failed: [...new Set(failed)] };
95
+ // A check with conflicting verdicts resolves to fail: the reader saw both.
96
+ for (const id of [...new Set(passed)]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
97
+ const undecided = new Set(known);
98
+ for (const id of [...passed, ...failed]) undecided.delete(id);
99
+ return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided] };
64
100
  }
65
101
 
66
- /** Pure decision core: given a parsed report, mutate the task tree with the
67
- * rollback set. Returns the outcome (rolled-back ids). Exported for tests. */
68
- export function applyAuditOutcome(
69
- checklist: CheckItem[],
70
- tasks: TaskView[],
71
- round: number,
72
- passed: string[],
73
- failed: string[],
74
- report: string,
75
- ): AuditOutcome {
76
- for (const id of passed) {
77
- const item = checklist.find((candidate) => candidate.id === id);
78
- if (item) item.done = true;
79
- }
80
- const rolledBack: string[] = [];
81
- for (const id of failed) {
82
- rolledBack.push(...auditRollbackSet(tasks, checklist, id));
83
- }
84
- return { round, passed, failed, rolledBack, report };
102
+ /** Pure decision core: classify a parsed report into the audit outcome. It
103
+ * touches neither the checklist nor the task tree — the caller owns writing
104
+ * `done` and computing the rollback set, so there is exactly one authority for
105
+ * both. Exported for tests. */
106
+ export function applyAuditOutcome(round: number, parsed: ParsedAudit, report: string): AuditOutcome {
107
+ return {
108
+ round,
109
+ passed: parsed.passed,
110
+ failed: parsed.failed,
111
+ undeterminable: parsed.undeterminable,
112
+ report,
113
+ };
85
114
  }
86
115
 
87
116
  /** Skipped-pass checks (all covered tasks skipped) pass without audit. */
@@ -89,9 +118,13 @@ export function presolvedCheckIds(checklist: CheckItem[], tasks: TaskView[]): st
89
118
  return skippedPassCheckIds(checklist, tasks);
90
119
  }
91
120
 
92
- /** Spawn the audit subagent and apply its outcome. Returns null when the
93
- * audit subagent itself failed to run (treated as a failed round with an
94
- * empty rollback; the caller counts it against the round cap). */
121
+ /** Spawn the audit subagent and classify its report. Returns `{ cancelled: true }`
122
+ * when the child was aborted (session shutdown, stop, restore, tree switch) —
123
+ * such a round burns no budget and sends no wake; `null` when the subagent
124
+ * itself failed to run (treated as an all-undeterminable committed round; the
125
+ * caller counts it against the round cap). */
126
+ export type AuditRoundResult = AuditOutcome | { cancelled: true } | null;
127
+
95
128
  export async function runCompletionAudit(
96
129
  ctx: ExtensionContext,
97
130
  opts: {
@@ -100,27 +133,95 @@ export async function runCompletionAudit(
100
133
  tasks: TaskView[];
101
134
  round: number;
102
135
  model?: string;
136
+ thinkingLevel?: string;
137
+ timeoutMs?: number;
103
138
  signal?: AbortSignal;
139
+ onProgress?: (event: import("./subagent.ts").SubagentProgressEvent) => void;
104
140
  },
105
- ): Promise<AuditOutcome | null> {
141
+ ): Promise<AuditRoundResult> {
106
142
  const { runPiSubagent } = await import("./subagent.ts");
143
+ const pending = auditablePendingChecks(opts.checklist, opts.tasks);
107
144
  const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round);
108
- let agentPrompt = "You are a read-only completion auditor for pi-plans.";
109
- try {
110
- agentPrompt = fs.readFileSync(new URL("../agents/reviewer.md", import.meta.url), "utf8");
111
- } catch {
112
- /* fall back to the inline prompt */
113
- }
145
+ // agents/auditor.md, not agents/reviewer.md: the reviewer prompt mandates a
146
+ // plan-review shape (`## Findings` / `## Questions`, F-###) and never says
147
+ // "verdict", so auditing under it produced reports this parser could not
148
+ // read at all -- every check then fell through to fail-closed.
149
+ // v0.8: a missing agent definition is a hard error — the silent inline
150
+ // fallback once swapped in a minimal prompt that produced unparsable
151
+ // reports and burned whole audit budgets as undeterminable.
152
+ const agentPrompt = fs.readFileSync(new URL("../agents/execution-reviewer.md", import.meta.url), "utf8");
114
153
  const result = await runPiSubagent({
115
- systemPrompt: `${agentPrompt}\n\nYou are acting as the completion auditor; the task brief below defines the output contract.`,
154
+ systemPrompt: agentPrompt,
116
155
  task,
117
156
  cwd: ctx.cwd,
118
157
  model: opts.model,
158
+ thinkingLevel: opts.thinkingLevel,
159
+ timeoutMs: opts.timeoutMs,
119
160
  tools: ["read", "grep", "find", "ls"],
120
161
  signal: opts.signal,
162
+ onProgress: opts.onProgress,
121
163
  });
122
- if (!result.ok) return null;
123
- const { passed, failed } = parseAuditReport(result.output, opts.checklist);
124
- messaging().appendEntry("pi-plans-audit", { planPath: opts.planPath, round: opts.round, passed, failed });
125
- return applyAuditOutcome(opts.checklist, opts.tasks, opts.round, passed, failed, result.output);
164
+ if (!result.ok) {
165
+ if (result.cancelled === true) return { cancelled: true };
166
+ return null;
167
+ }
168
+ const parsed = parseAuditReport(result.output, pending.map((item) => item.id));
169
+ messaging().appendEntry("pi-plans-audit", {
170
+ planPath: opts.planPath,
171
+ round: opts.round,
172
+ passed: parsed.passed,
173
+ failed: parsed.failed,
174
+ undeterminable: parsed.undeterminable,
175
+ });
176
+ return applyAuditOutcome(opts.round, parsed, result.output);
177
+ }
178
+
179
+ /** Persist one review-round attempt under `<run-dir>/execution-review/`.
180
+ * The attempt index (not the budget round) names the file, so a discarded
181
+ * attempt's evidence survives its re-run: round-<budgetRound>-attempt-<k>.md.
182
+ * These files are the only cross-session record of what a round saw — the
183
+ * in-flight marker itself is memory-only. Best-effort: an unwritable run dir
184
+ * must never fail the loop itself. */
185
+ export function writeReviewRoundReport(
186
+ runDir: string,
187
+ entry: {
188
+ budgetRound: number;
189
+ attempt: number;
190
+ outcome: "passed" | "failed" | "undeterminable" | "discarded" | "spawn-failed" | "cancelled";
191
+ passed: string[];
192
+ failed: string[];
193
+ undeterminable: string[];
194
+ discardedReason?: string;
195
+ fingerprintCaptured?: string;
196
+ fingerprintFound?: string;
197
+ coveredTaskIds: string[];
198
+ report: string;
199
+ },
200
+ ): string | null {
201
+ try {
202
+ const dir = path.join(runDir, "execution-review");
203
+ fs.mkdirSync(dir, { recursive: true });
204
+ const file = path.join(dir, `round-${entry.budgetRound}-attempt-${entry.attempt}.md`);
205
+ const lines = [
206
+ `# Execution review round ${entry.budgetRound} (attempt ${entry.attempt})`,
207
+ "",
208
+ `- outcome: ${entry.outcome}`,
209
+ `- passed: ${entry.passed.join(", ") || "(none)"}`,
210
+ `- failed: ${entry.failed.join(", ") || "(none)"}`,
211
+ `- undeterminable: ${entry.undeterminable.join(", ") || "(none)"}`,
212
+ ...(entry.discardedReason ? [`- discarded: ${entry.discardedReason}`] : []),
213
+ `- fingerprint (captured): ${entry.fingerprintCaptured ?? "(n/a)"}`,
214
+ `- fingerprint (at resolve): ${entry.fingerprintFound ?? "(n/a)"}`,
215
+ `- covered tasks: ${entry.coveredTaskIds.join(", ") || "(none)"}`,
216
+ "",
217
+ "## Report",
218
+ "",
219
+ entry.report,
220
+ "",
221
+ ];
222
+ fs.writeFileSync(file, lines.join("\n"), "utf8");
223
+ return file;
224
+ } catch {
225
+ return null;
226
+ }
126
227
  }
@@ -284,7 +284,12 @@ export async function applyGraphCommand(args: string, ctx: CommandContext): Prom
284
284
  errors > 0 ? "error" : "info",
285
285
  );
286
286
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
287
- if (active) setRunStatus(ctx.cwd, active.run_id, "executing");
287
+ if (active) {
288
+ // Never demote a verifying run back to executing: the execution-review
289
+ // loop owns the status until it converges (or pauses at the round cap).
290
+ const current = getRun(ctx.cwd, active.run_id)?.status;
291
+ if (current !== "verifying") setRunStatus(ctx.cwd, active.run_id, "executing");
292
+ }
288
293
  }
289
294
 
290
295
  export async function graphStatusCommand(_args: string, ctx: CommandContext): Promise<void> {
package/src/dashboard.ts CHANGED
@@ -16,6 +16,7 @@
16
16
  */
17
17
 
18
18
  import type { CheckItem } from "./plan.ts";
19
+ import { REVIEW_MAX_ROUNDS } from "./auditor.ts";
19
20
  import { truncateToWidth, visibleWidth } from "./refine-ui-helpers.ts";
20
21
  import {
21
22
  allTasksTerminal,
@@ -38,15 +39,24 @@ export interface DashboardModel {
38
39
  /** Audit round counter; null = audit not yet started. */
39
40
  auditRounds: number | null;
40
41
  auditFailed: string[];
42
+ /** Checks whose verdict the auditor could not be read for. Neither passed
43
+ * nor failed: shown so the run does not read as finished. */
44
+ auditUndeterminable: string[];
45
+ /** v0.8: true while an execution-review round is in flight. The panel must
46
+ * never render `audit complete ✓` while this is set — the round-1 mis-cue
47
+ * showed the tick for the whole duration of a running audit. */
48
+ reviewRunning: boolean;
41
49
  startedAt: string;
42
50
  usage: { inToks: number; outToks: number };
43
51
  }
44
52
 
45
- /** Tree marker for one task row. */
53
+ /** Tree marker for one task row. A rolled-back task is `pending` but still
54
+ * carries the evidence of its previous attempt, so it gets its own marker
55
+ * rather than reading as untouched work. */
46
56
  export function taskMarker(task: TaskView, currentId: string | null): string {
47
57
  if (taskIsTerminal(task)) return task.status === "skipped" ? "~" : "✓";
48
58
  if (task.id === currentId) return "▸";
49
- return "·";
59
+ return task.evidence === undefined ? "·" : "↺";
50
60
  }
51
61
 
52
62
  /** Truncate to at most `max` display columns, ellipsis when clipped. */
@@ -157,16 +167,39 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
157
167
  lines.push(boxRow("│", ` ${clip(cur.files.join(", "), inner - 4)}`, " ", width));
158
168
  }
159
169
  } else if (allTasksTerminal(model.tasks)) {
160
- const auditLine = model.auditFailed.length > 0
161
- ? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
162
- : model.auditRounds !== null
163
- ? "audit complete ✓"
164
- : "all tasks terminal — audit pending";
170
+ // v0.8 mis-cue guard: `audit complete ✓` ONLY when nothing is running
171
+ // and nothing is owed. A running or owed review shows its round number
172
+ // instead — the panel must look alive, never finished-then-silent.
173
+ const owed = model.checklist.some((item) => !item.done);
174
+ const nextRound = (model.auditRounds ?? 0) + 1;
175
+ const auditLine = model.reviewRunning
176
+ ? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} running — read-only reviewer verifying`
177
+ : model.auditFailed.length > 0
178
+ ? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
179
+ : model.auditUndeterminable.length > 0
180
+ ? `audit: ${model.auditUndeterminable.length} check(s) undeterminable — verdict unreadable`
181
+ : owed
182
+ ? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — verdict pending`
183
+ : model.auditRounds !== null
184
+ ? "audit complete ✓"
185
+ : "all tasks terminal — audit pending";
165
186
  lines.push(boxRow("│", ` ${clip(auditLine, inner - 2)}`, " ", width));
166
187
  }
167
188
  if (!narrow && model.auditFailed.length > 0) {
168
189
  lines.push(boxRow("│", ` ✗ ${clip(model.auditFailed.join(", "), inner - 3)}`, " ", width));
169
190
  }
191
+ if (!narrow && model.auditUndeterminable.length > 0) {
192
+ lines.push(boxRow("│", ` ? ${clip(model.auditUndeterminable.join(", "), inner - 3)}`, " ", width));
193
+ }
194
+ // Rolled-back work keeps the evidence of the attempt that was rolled back.
195
+ // The tree view shows it per row; the compact panel has room for a summary
196
+ // and must show it too, otherwise the retention is invisible outside the
197
+ // expanded view. Capped at two rows to protect the fixed panel height.
198
+ for (const rolled of flattenTaskViews(model.tasks)
199
+ .filter((task) => task.status === "pending" && task.evidence !== undefined)
200
+ .slice(0, 2)) {
201
+ lines.push(boxRow("│", ` ↺ ${rolled.id} ${clip(rolled.evidence!, inner - 6)}`, " ", width));
202
+ }
170
203
  lines.push(boxRow("└", "─ Ctrl+Shift+T tree", "─", width, "┘"));
171
204
  // The title row carries its own colours; every other row is muted.
172
205
  const painted = theme ? lines.map((line, i) => (i === 0 ? line : theme.fg("muted", line))) : lines;
@@ -183,7 +216,7 @@ export function deriveDashboardModel(
183
216
  topic: string,
184
217
  tasks: TaskView[],
185
218
  checklist: CheckItem[],
186
- extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; startedAt?: string; usage?: { inToks: number; outToks: number } },
219
+ extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; startedAt?: string; usage?: { inToks: number; outToks: number } },
187
220
  ): DashboardModel {
188
221
  return {
189
222
  topic,
@@ -193,6 +226,8 @@ export function deriveDashboardModel(
193
226
  pausedReason: extra?.pausedReason,
194
227
  auditRounds: extra?.auditRounds ?? null,
195
228
  auditFailed: extra?.auditFailed ?? [],
229
+ auditUndeterminable: extra?.auditUndeterminable ?? [],
230
+ reviewRunning: extra?.reviewRunning ?? false,
196
231
  startedAt: extra?.startedAt ?? new Date().toISOString(),
197
232
  usage: extra?.usage ?? { inToks: 0, outToks: 0 },
198
233
  };
@@ -226,6 +261,9 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
226
261
  if (wide && task.deps.length > 0) line += ` (deps: ${task.deps.join(", ")})`;
227
262
  if (task.status === "skipped" && task.skipReason) line += ` ~${task.skipReason}`;
228
263
  if (task.status === "complete" && task.evidence) line += ` ✓${clip(task.evidence, 40)}`;
264
+ // Rolled-back work keeps its evidence, and that evidence is the only
265
+ // record of what the previous attempt did.
266
+ if (task.status === "pending" && task.evidence) line += ` ↺${clip(task.evidence, 40)}`;
229
267
  lines.push(line);
230
268
  for (const child of task.children) row(child, depth + 1);
231
269
  }; for (const task of model.tasks) row(task, 0);
@@ -235,9 +273,17 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
235
273
  const mark = model.auditFailed.includes(item.id) ? "✗" : item.done ? "☑" : "☐";
236
274
  lines.push(` ${mark} ${item.id}${wide ? ` ${clip(item.text.split(";")[1] ?? item.text, Math.min(80, width))}` : ""}`);
237
275
  }
238
- if (model.auditRounds !== null) {
276
+ if (model.reviewRunning) {
277
+ lines.push("");
278
+ lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${REVIEW_MAX_ROUNDS} running`);
279
+ } else if (model.auditRounds !== null) {
239
280
  lines.push("");
240
- lines.push(`Completion audit: round ${model.auditRounds}${model.auditFailed.length > 0 ? ` — failed: ${model.auditFailed.join(", ")}` : " — passed ✓"}`);
281
+ const verdict = model.auditFailed.length > 0
282
+ ? ` — failed: ${model.auditFailed.join(", ")}`
283
+ : model.auditUndeterminable.length > 0
284
+ ? ` — undeterminable: ${model.auditUndeterminable.join(", ")}`
285
+ : " — passed ✓";
286
+ lines.push(`Execution review: round ${model.auditRounds}${verdict}`);
241
287
  }
242
288
  const painted = theme ? lines.map((line) => theme.fg("muted", line)) : lines;
243
289
  return clampLines(painted, width);