pi-plans 0.7.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +5 -12
  3. package/README.md +5 -5
  4. package/agents/execution-reviewer.md +92 -0
  5. package/index.ts +13 -23
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +14 -7
  8. package/references/plan-artifact-template.md +11 -1
  9. package/references/state-and-config.md +3 -3
  10. package/scripts/validate.ts +20 -3
  11. package/src/auditor.ts +306 -63
  12. package/src/code-graph/commands.ts +6 -1
  13. package/src/dashboard.ts +91 -13
  14. package/src/exec.ts +835 -142
  15. package/src/plan.ts +1 -1
  16. package/src/refine-ui-state.ts +1 -1
  17. package/src/refine-ui.ts +40 -8
  18. package/src/resume-command.ts +19 -3
  19. package/src/resume.ts +5 -1
  20. package/src/staleness.ts +53 -0
  21. package/src/state.ts +1 -0
  22. package/src/task-tool.ts +1 -1
  23. package/src/tasks.ts +62 -5
  24. package/src/ui-language.ts +4 -0
  25. package/src/workflow-state.ts +93 -6
  26. package/tests/analyze-refs.test.ts +1 -1
  27. package/tests/auditor.test.ts +299 -16
  28. package/tests/dashboard.test.ts +202 -2
  29. package/tests/exec-review-loop.test.ts +724 -0
  30. package/tests/exec.test.ts +198 -44
  31. package/tests/extension-load.test.ts +1 -1
  32. package/tests/refine-ui.test.ts +25 -2
  33. package/tests/resume-lifecycle.test.ts +5 -1
  34. package/tests/resume.test.ts +6 -0
  35. package/tests/staleness.test.ts +76 -0
  36. package/tests/state.test.ts +4 -0
  37. package/tests/tasks.test.ts +142 -0
  38. package/tests/workflow-state.test.ts +105 -0
  39. package/tools/analyze-refs.ts +17 -6
  40. package/tools/execute-plan.ts +12 -5
  41. package/tools/plans.ts +1 -1
  42. package/tools/refine.ts +22 -3
package/src/auditor.ts CHANGED
@@ -1,87 +1,249 @@
1
1
  /**
2
- * Completion auditor (v0.6.1): when every task reaches a terminal state, an
3
- * independent read-only subagent verifies the plan's verification checks
4
- * against the worktree. Failed checks roll their covered tasks back to
5
- * pending (exclusively inside the audit flow); three failed rounds pause the
6
- * run for the user — or terminate it as stopped under auto-approve/headless
7
- * so pipelines never hang.
2
+ * Execution reviewer (v0.8): when every task
3
+ * reaches a terminal state, an independent read-only subagent verifies the
4
+ * plan's verification checks against the worktree. Failed checks roll their
5
+ * covered tasks back to pending (exclusively inside the review flow). The
6
+ * loop is bounded at REVIEW_MAX_ROUNDS committed rounds; exhaustion pauses
7
+ * the run for the user in every mode (fail-closed, never a hang and never a
8
+ * silent stop) — only an explicit /plans-execute confirmation grants a fresh
9
+ * budget.
8
10
  */
9
11
 
10
12
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
11
13
  import * as fs from "node:fs";
12
- import { auditRollbackSet, auditableChecks, skippedPassCheckIds, type TaskView } from "./tasks.ts";
13
- import type { CheckItem } from "./plan.ts";
14
+ import * as path from "node:path";
15
+ import { auditableChecks, flattenTaskViews, skippedPassCheckIds, type TaskView } from "./tasks.ts";
16
+ import { normalizeTaskId, type CheckItem } from "./plan.ts";
14
17
  import { messaging } from "./messaging.ts";
15
18
 
16
- export const AUDIT_MAX_ROUNDS = 3;
19
+ /** Budget cap: committed rounds per user-granted budget. Discarded (fingerprint-changed) attempts do not count. */
20
+ export const REVIEW_MAX_ROUNDS = 5;
21
+
22
+ /** Legacy alias for one release: checkpoints and old builds still know this name. */
23
+ export const AUDIT_MAX_ROUNDS = REVIEW_MAX_ROUNDS;
17
24
 
18
25
  export interface AuditOutcome {
19
26
  round: number;
20
27
  passed: string[];
21
28
  failed: string[];
22
- /** Rolled-back task ids (empty when the audit passed). */
23
- rolledBack: string[];
29
+ /** Checks whose report carried no readable verdict, or whose evidence was
30
+ * inconclusive. Never treated as `failed`: see src/exec.ts. */
31
+ undeterminable: string[];
32
+ /** Severity-graded implementation findings from this round (v0.9). Absent
33
+ * on legacy shapes means "no findings reported"; the parse boundary always
34
+ * sets a concrete array so hand-written construction sites cannot silently
35
+ * drop them. */
36
+ findings?: ReviewFinding[];
24
37
  report: string;
25
38
  }
26
39
 
40
+ /** Severity vocabulary for implementation findings; mirrors the plan
41
+ * priority words. `malformed` marks a bullet the grammar parser could not
42
+ * read — it is recorded and displayed but never drives a rollback. */
43
+ export type FindingSeverity = "high" | "medium" | "low" | "malformed";
44
+
45
+ export interface ReviewFinding {
46
+ /** Stable id, normalized uppercase (`F-001`). Reused verbatim across
47
+ * rounds while the problem persists; absence from the newest round's
48
+ * report is the resolution signal. */
49
+ id: string;
50
+ severity: FindingSeverity;
51
+ /** Task ids owning the defect (normalized), `[]` when unmapped. */
52
+ taskIds: string[];
53
+ /** Required for unmapped high findings: the runner appends this as a new
54
+ * plan task mechanically, so it must be a self-contained imperative title. */
55
+ proposedTask?: string;
56
+ note: string;
57
+ evidence: string;
58
+ /** Original bullet, for degraded records and round reports. */
59
+ raw: string;
60
+ }
61
+
62
+ /** Parse the verdict lines of an audit report against the checks that are
63
+ * still pending. A pending check with no readable verdict is
64
+ * `undeterminable`, never `failed` — the caller must be able to tell "the
65
+ * work is wrong" apart from "I could not read the answer". */
66
+ export interface ParsedAudit {
67
+ passed: string[];
68
+ failed: string[];
69
+ undeterminable: string[];
70
+ findings: ReviewFinding[];
71
+ }
72
+
73
+ /** One finding bullet's field-capture helper: lazily up to the next
74
+ * `; <known-field>:` boundary or the end of the line, so free text in one
75
+ * field cannot swallow the next. */
76
+ function captureField(line: string, field: string): string | undefined {
77
+ const m = line.match(new RegExp(`${field}:\\s*(.*?)(?=;\\s*(?:severity|tasks|proposed-task|note|evidence):|$)`, "i"));
78
+ return m ? m[1].trim().replace(/^[*_`~]+|[*_`~]+$/g, "") : undefined;
79
+ }
80
+
81
+ /** Parse the findings bullets of a review report. Degrade, never crash, never
82
+ * roll back: a bullet with an F-### id but unreadable fields is recorded with
83
+ * severity "malformed" (visible, non-blocking), exactly like the verdict
84
+ * parser's emphasis tolerance — a strict grammar once silently discarded
85
+ * verdicts and fail-closed rolled correct work back. */
86
+ export function parseFindings(report: string, knownTaskIds?: Set<string>): ReviewFinding[] {
87
+ const findings: ReviewFinding[] = [];
88
+ const seen = new Set<string>();
89
+ for (const match of report.matchAll(/^\s*[-*]\s+`?(F-\d+)`?\b/gim)) {
90
+ const id = match[1].toUpperCase();
91
+ if (seen.has(id)) continue; // conflicting duplicates resolve to the first
92
+ seen.add(id);
93
+ // The bullet regex's leading \s* may swallow the preceding newline, so
94
+ // slice(index) can start mid-whitespace; strip it before taking the line.
95
+ const line = match.input!.slice(match.index!).replace(/^[\s]+/, "").split(/\n/)[0];
96
+ const severityRaw = captureField(line, "severity")?.toLowerCase();
97
+ const tasksRaw = captureField(line, "tasks");
98
+ const taskIds = (tasksRaw ?? "")
99
+ .split(/[,,,、]/)
100
+ .map((token) => normalizeTaskId(token.trim()))
101
+ .filter((t): t is string => t !== null)
102
+ .filter((t) => !knownTaskIds || knownTaskIds.has(t));
103
+ const severity: FindingSeverity =
104
+ severityRaw === "high" || severityRaw === "medium" || severityRaw === "low" ? severityRaw : "malformed";
105
+ if (severity === "malformed" || !tasksRaw) {
106
+ // Unreadable severity, or the grammar's mandatory tasks field missing:
107
+ // degrade to a recorded non-blocking entry.
108
+ findings.push({ id, severity: "malformed", taskIds: [], note: captureField(line, "note") ?? line.trim(), evidence: captureField(line, "evidence") ?? "", raw: line.trim() });
109
+ continue;
110
+ }
111
+ findings.push({
112
+ id,
113
+ severity,
114
+ taskIds,
115
+ proposedTask: captureField(line, "proposed-task") || undefined,
116
+ note: captureField(line, "note") ?? "",
117
+ evidence: captureField(line, "evidence") ?? "",
118
+ raw: line.trim(),
119
+ });
120
+ }
121
+ return findings;
122
+ }
123
+
124
+ /** Checks this round must judge: auditable (covering at least one task) and
125
+ * not already done. The brief and the coverage self-check both use this, so a
126
+ * well-formed round-2 report that omits an already-done check is not mistaken
127
+ * for a contract violation. */
128
+ export function auditablePendingChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
129
+ return auditableChecks(checklist, tasks).filter((item) => !item.done);
130
+ }
131
+
27
132
  /** Build the audit brief for the read-only subagent. Exported for tests. */
28
- export function buildAuditTask(planPath: string, checklist: CheckItem[], tasks: TaskView[], round: number): string {
29
- const checks = auditableChecks(checklist, tasks)
133
+ export function buildAuditTask(
134
+ planPath: string,
135
+ checklist: CheckItem[],
136
+ tasks: TaskView[],
137
+ round: number,
138
+ priorFindings: ReviewFinding[] = [],
139
+ ): string {
140
+ const checks = auditablePendingChecks(checklist, tasks)
30
141
  .map((check) => `- \`${check.id}\`: ${check.text}`)
31
142
  .join("\n");
32
- return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks.
143
+ const taskList = flattenTaskViews(tasks)
144
+ .map((t) => `- \`${t.id}\`: ${t.title} (${t.status})`)
145
+ .join("\n");
146
+ const prior = priorFindings.length
147
+ ? `Unresolved findings from earlier rounds (reuse these exact ids while the problem persists; a problem is resolved only by no longer reporting it):
148
+
149
+ ${priorFindings.map((f) => `- \`${f.id}\` — severity: ${f.severity}; tasks: ${f.taskIds.join(", ") || "none"}; note: ${f.note}`).join("\n")}`
150
+ : "(none — this is the first round with findings in scope)";
151
+ return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks, and report implementation findings that drive the fix loop.
33
152
 
34
- Target plan: ${planPath} (audit round ${round})
153
+ Target plan: ${planPath} (review round ${round})
35
154
 
36
155
  Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
37
156
 
38
157
  Evidence: inspect the repository with read, grep, find, ls, and targeted commands (bash is not granted — rely on the read tools) before judging each check. Tests may be referenced from their recorded evidence; do not re-run them.
39
158
 
40
- Checks to verify (only these; checks covering no task are excluded):
159
+ Checks to verify (only these; checks covering no task and checks already satisfied in an earlier round are excluded):
41
160
 
42
161
  ${checks}
43
162
 
44
- Output: Markdown with exactly one section per check, in checklist order:
163
+ Plan tasks (the only ids valid in a finding's tasks field):
164
+
165
+ ${taskList}
166
+
167
+ ${prior}
45
168
 
46
- - \`VC-###\` — verdict: pass | fail; evidence: <repo path/command proving it>; note: <one line>.
169
+ Output: exactly two sections, in this order.
47
170
 
48
- Every check needs a verdict backed by evidence you actually inspected. If the evidence is inconclusive, verdict is fail with what is missing.`;
171
+ 1. Verification verdicts — one section per check, in the order above:
172
+
173
+ - \`VC-###\` — verdict: pass | fail | undeterminable; evidence: <repo path/command proving it>; note: <one line>.
174
+
175
+ Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence.
176
+
177
+ 2. Implementation findings — one bullet per defect anywhere in the implemented change (not only what the checks cover), exact line grammar:
178
+
179
+ - \`F-###\` — severity: high | medium | low; tasks: Task-N, Task-M | none; proposed-task: <imperative one-line title>; note: <one line>; evidence: <repo path or quoted excerpt>
180
+
181
+ \`high\` wakes the executor for a fix round; \`medium\`/\`low\` are recorded. When no existing task owns the defect use \`tasks: none\` and (required for high) a self-contained \`proposed-task:\` title. If nothing is worth reporting, emit exactly \`- none.\` under the findings heading.`;
49
182
  }
50
183
 
51
184
  /** Parse the audit subagent's verdict lines. Exported for tests. */
52
- export function parseAuditReport(report: string, checklist: CheckItem[]): { passed: string[]; failed: string[] } {
185
+ export function parseAuditReport(report: string, pendingVcIds: string[], knownTaskIds?: Set<string>): ParsedAudit {
53
186
  const passed: string[] = [];
54
187
  const failed: string[] = [];
55
- const known = new Set(checklist.map((item) => item.id));
56
- for (const match of report.matchAll(/`?(VC-\d+)`?[^\n]*?verdict:\s*(pass|fail)/gi)) {
57
- const id = match[1].toUpperCase();
58
- if (!known.has(id)) continue;
59
- (match[2].toLowerCase() === "pass" ? passed : failed).push(id);
188
+ const known = new Set(pendingVcIds.map((id) => id.toUpperCase()));
189
+ // The reviewer writes Markdown, so a verdict may carry emphasis
190
+ // (`**pass**`, `*pass*`, `_pass_`, `` `pass` ``). Requiring a bare token
191
+ // silently discarded such verdicts, and the caller's fail-closed rule then
192
+ // marked every check failed and rolled the whole run back -- reporting
193
+ // correct work as failure. Tolerate the markers; \b keeps `passed` and
194
+ // `passing` from matching.
195
+ // v0.9: finding bullets (F-###) are excluded from verdict scanning — a
196
+ // finding's note may cite a VC id, and that must never register a verdict.
197
+ // v0.9.1 (F-011): the contract promises "one section per check" and an
198
+ // equally literal reading puts the id in a `### VC-###` heading with the
199
+ // verdict on a line of its own below it. The old same-line-only scan
200
+ // parsed such reports to ZERO verdicts and burned whole budgets as
201
+ // undeterminable. The scan is now section-aware: a line that NAMES a
202
+ // known check at a heading/bullet start opens that check's section, and a
203
+ // bare `verdict:` token attributes to the nearest open section; an
204
+ // id-and-verdict pair on one line stays direct.
205
+ const record = (id: string, verdict: string): void => {
206
+ if (verdict === "pass") passed.push(id);
207
+ else if (verdict === "fail") failed.push(id);
208
+ };
209
+ const verdictToken = /verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i;
210
+ let current: string | null = null;
211
+ for (const line of report.split(/\n/)) {
212
+ if (/^\s*[-*]\s+`?F-\d+`?\b/i.test(line)) continue; // finding bullet
213
+ const sectionId = line.match(/^\s*(?:#{1,6}\s+|[-*]\s+)?`?(VC-\d+)`?\b/i);
214
+ if (sectionId) {
215
+ const id = sectionId[1].toUpperCase();
216
+ if (known.has(id)) current = id;
217
+ }
218
+ const direct = line.match(/`?(VC-\d+)`?[^\n]*?verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i);
219
+ if (direct) {
220
+ const id = direct[1].toUpperCase();
221
+ if (known.has(id)) record(id, direct[2].toLowerCase());
222
+ continue;
223
+ }
224
+ const bare = line.match(verdictToken);
225
+ if (bare && current) record(current, bare[1].toLowerCase());
60
226
  }
61
- // Dedupe, keep first occurrence order; a check with conflicting verdicts fails.
62
- for (const id of [...passed]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
63
- return { passed: [...new Set(passed)], failed: [...new Set(failed)] };
227
+ // A check with conflicting verdicts resolves to fail: the reader saw both.
228
+ for (const id of [...new Set(passed)]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
229
+ const undecided = new Set(known);
230
+ for (const id of [...passed, ...failed]) undecided.delete(id);
231
+ return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided], findings: parseFindings(report, knownTaskIds) };
64
232
  }
65
233
 
66
- /** Pure decision core: given a parsed report, mutate the task tree with the
67
- * rollback set. Returns the outcome (rolled-back ids). Exported for tests. */
68
- export function applyAuditOutcome(
69
- checklist: CheckItem[],
70
- tasks: TaskView[],
71
- round: number,
72
- passed: string[],
73
- failed: string[],
74
- report: string,
75
- ): AuditOutcome {
76
- for (const id of passed) {
77
- const item = checklist.find((candidate) => candidate.id === id);
78
- if (item) item.done = true;
79
- }
80
- const rolledBack: string[] = [];
81
- for (const id of failed) {
82
- rolledBack.push(...auditRollbackSet(tasks, checklist, id));
83
- }
84
- return { round, passed, failed, rolledBack, report };
234
+ /** Pure decision core: classify a parsed report into the audit outcome. It
235
+ * touches neither the checklist nor the task tree — the caller owns writing
236
+ * `done` and computing the rollback set, so there is exactly one authority for
237
+ * both. Exported for tests. */
238
+ export function applyAuditOutcome(round: number, parsed: ParsedAudit, report: string): AuditOutcome {
239
+ return {
240
+ round,
241
+ passed: parsed.passed,
242
+ failed: parsed.failed,
243
+ undeterminable: parsed.undeterminable,
244
+ findings: parsed.findings,
245
+ report,
246
+ };
85
247
  }
86
248
 
87
249
  /** Skipped-pass checks (all covered tasks skipped) pass without audit. */
@@ -89,9 +251,13 @@ export function presolvedCheckIds(checklist: CheckItem[], tasks: TaskView[]): st
89
251
  return skippedPassCheckIds(checklist, tasks);
90
252
  }
91
253
 
92
- /** Spawn the audit subagent and apply its outcome. Returns null when the
93
- * audit subagent itself failed to run (treated as a failed round with an
94
- * empty rollback; the caller counts it against the round cap). */
254
+ /** Spawn the audit subagent and classify its report. Returns `{ cancelled: true }`
255
+ * when the child was aborted (session shutdown, stop, restore, tree switch) —
256
+ * such a round burns no budget and sends no wake; `null` when the subagent
257
+ * itself failed to run (treated as an all-undeterminable committed round; the
258
+ * caller counts it against the round cap). */
259
+ export type AuditRoundResult = AuditOutcome | { cancelled: true } | null;
260
+
95
261
  export async function runCompletionAudit(
96
262
  ctx: ExtensionContext,
97
263
  opts: {
@@ -99,28 +265,105 @@ export async function runCompletionAudit(
99
265
  checklist: CheckItem[];
100
266
  tasks: TaskView[];
101
267
  round: number;
268
+ /** Unresolved findings from earlier committed rounds, injected into
269
+ * the brief so the reviewer reuses stable ids (v0.9). */
270
+ priorFindings?: ReviewFinding[];
102
271
  model?: string;
272
+ thinkingLevel?: string;
273
+ timeoutMs?: number;
103
274
  signal?: AbortSignal;
275
+ onProgress?: (event: import("./subagent.ts").SubagentProgressEvent) => void;
104
276
  },
105
- ): Promise<AuditOutcome | null> {
277
+ ): Promise<AuditRoundResult> {
106
278
  const { runPiSubagent } = await import("./subagent.ts");
107
- const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round);
108
- let agentPrompt = "You are a read-only completion auditor for pi-plans.";
109
- try {
110
- agentPrompt = fs.readFileSync(new URL("../agents/reviewer.md", import.meta.url), "utf8");
111
- } catch {
112
- /* fall back to the inline prompt */
113
- }
279
+ const pending = auditablePendingChecks(opts.checklist, opts.tasks);
280
+ const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round, opts.priorFindings ?? []);
281
+ // agents/auditor.md, not agents/reviewer.md: the reviewer prompt mandates a
282
+ // plan-review shape (`## Findings` / `## Questions`, F-###) and never says
283
+ // "verdict", so auditing under it produced reports this parser could not
284
+ // read at all -- every check then fell through to fail-closed.
285
+ // v0.8: a missing agent definition is a hard error — the silent inline
286
+ // fallback once swapped in a minimal prompt that produced unparsable
287
+ // reports and burned whole audit budgets as undeterminable.
288
+ const agentPrompt = fs.readFileSync(new URL("../agents/execution-reviewer.md", import.meta.url), "utf8");
114
289
  const result = await runPiSubagent({
115
- systemPrompt: `${agentPrompt}\n\nYou are acting as the completion auditor; the task brief below defines the output contract.`,
290
+ systemPrompt: agentPrompt,
116
291
  task,
117
292
  cwd: ctx.cwd,
118
293
  model: opts.model,
294
+ thinkingLevel: opts.thinkingLevel,
295
+ timeoutMs: opts.timeoutMs,
119
296
  tools: ["read", "grep", "find", "ls"],
120
297
  signal: opts.signal,
298
+ onProgress: opts.onProgress,
121
299
  });
122
- if (!result.ok) return null;
123
- const { passed, failed } = parseAuditReport(result.output, opts.checklist);
124
- messaging().appendEntry("pi-plans-audit", { planPath: opts.planPath, round: opts.round, passed, failed });
125
- return applyAuditOutcome(opts.checklist, opts.tasks, opts.round, passed, failed, result.output);
300
+ if (!result.ok) {
301
+ if (result.cancelled === true) return { cancelled: true };
302
+ return null;
303
+ }
304
+ const parsed = parseAuditReport(result.output, pending.map((item) => item.id), new Set(flattenTaskViews(opts.tasks).map((t) => t.id)));
305
+ messaging().appendEntry("pi-plans-audit", {
306
+ planPath: opts.planPath,
307
+ round: opts.round,
308
+ passed: parsed.passed,
309
+ failed: parsed.failed,
310
+ undeterminable: parsed.undeterminable,
311
+ highFindings: parsed.findings.filter((f) => f.severity === "high").map((f) => f.id),
312
+ });
313
+ return applyAuditOutcome(opts.round, parsed, result.output);
314
+ }
315
+
316
+ /** Persist one review-round attempt under `<run-dir>/execution-review/`.
317
+ * The attempt index (not the budget round) names the file, so a discarded
318
+ * attempt's evidence survives its re-run: round-<budgetRound>-attempt-<k>.md.
319
+ * These files are the only cross-session record of what a round saw — the
320
+ * in-flight marker itself is memory-only. Best-effort: an unwritable run dir
321
+ * must never fail the loop itself. */
322
+ export function writeReviewRoundReport(
323
+ runDir: string,
324
+ entry: {
325
+ budgetRound: number;
326
+ attempt: number;
327
+ outcome: "passed" | "failed" | "undeterminable" | "discarded" | "spawn-failed" | "cancelled";
328
+ passed: string[];
329
+ failed: string[];
330
+ undeterminable: string[];
331
+ /** v0.9 findings from this round; recorded in the round report so the
332
+ * file stays the full evidence record. */
333
+ findings?: ReviewFinding[];
334
+ discardedReason?: string;
335
+ fingerprintCaptured?: string;
336
+ fingerprintFound?: string;
337
+ coveredTaskIds: string[];
338
+ report: string;
339
+ },
340
+ ): string | null {
341
+ try {
342
+ const dir = path.join(runDir, "execution-review");
343
+ fs.mkdirSync(dir, { recursive: true });
344
+ const file = path.join(dir, `round-${entry.budgetRound}-attempt-${entry.attempt}.md`);
345
+ const lines = [
346
+ `# Execution review round ${entry.budgetRound} (attempt ${entry.attempt})`,
347
+ "",
348
+ `- outcome: ${entry.outcome}`,
349
+ `- passed: ${entry.passed.join(", ") || "(none)"}`,
350
+ `- failed: ${entry.failed.join(", ") || "(none)"}`,
351
+ `- undeterminable: ${entry.undeterminable.join(", ") || "(none)"}`,
352
+ `- high findings: ${entry.findings?.filter((f) => f.severity === "high").map((f) => f.id).join(", ") || "(none)"}`,
353
+ ...(entry.findings?.length ? [`- findings: ${entry.findings.map((f) => `${f.id} (${f.severity}${f.proposedTask ? "; proposed: " + f.proposedTask : ""})`).join(" | ")}`] : []),
354
+ ...(entry.discardedReason ? [`- discarded: ${entry.discardedReason}`] : []),
355
+ `- fingerprint (captured): ${entry.fingerprintCaptured ?? "(n/a)"}`,
356
+ `- fingerprint (at resolve): ${entry.fingerprintFound ?? "(n/a)"}`,
357
+ `- covered tasks: ${entry.coveredTaskIds.join(", ") || "(none)"}`,
358
+ "",
359
+ "## Report",
360
+ "",
361
+ entry.report,
362
+ "",
363
+ ];
364
+ fs.writeFileSync(file, lines.join("\n"), "utf8");
365
+ return file;
366
+ } catch {
367
+ return null;
368
+ }
126
369
  }
@@ -284,7 +284,12 @@ export async function applyGraphCommand(args: string, ctx: CommandContext): Prom
284
284
  errors > 0 ? "error" : "info",
285
285
  );
286
286
  const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
287
- if (active) setRunStatus(ctx.cwd, active.run_id, "executing");
287
+ if (active) {
288
+ // Never demote a verifying run back to executing: the execution-review
289
+ // loop owns the status until it converges (or pauses at the round cap).
290
+ const current = getRun(ctx.cwd, active.run_id)?.status;
291
+ if (current !== "verifying") setRunStatus(ctx.cwd, active.run_id, "executing");
292
+ }
288
293
  }
289
294
 
290
295
  export async function graphStatusCommand(_args: string, ctx: CommandContext): Promise<void> {
package/src/dashboard.ts CHANGED
@@ -16,6 +16,7 @@
16
16
  */
17
17
 
18
18
  import type { CheckItem } from "./plan.ts";
19
+ import { REVIEW_MAX_ROUNDS } from "./auditor.ts";
19
20
  import { truncateToWidth, visibleWidth } from "./refine-ui-helpers.ts";
20
21
  import {
21
22
  allTasksTerminal,
@@ -38,15 +39,27 @@ export interface DashboardModel {
38
39
  /** Audit round counter; null = audit not yet started. */
39
40
  auditRounds: number | null;
40
41
  auditFailed: string[];
42
+ /** Checks whose verdict the auditor could not be read for. Neither passed
43
+ * nor failed: shown so the run does not read as finished. */
44
+ auditUndeterminable: string[];
45
+ /** v0.8: true while an execution-review round is in flight. The panel must
46
+ * never render `audit complete ✓` while this is set — the round-1 mis-cue
47
+ * showed the tick for the whole duration of a running audit. */
48
+ reviewRunning: boolean;
49
+ /** v0.9: unresolved findings from the newest committed review round
50
+ * (stable ids). High entries block completion; the rest are recorded. */
51
+ findings: Array<{ id: string; severity: string; note: string; taskIds: string[] }>;
41
52
  startedAt: string;
42
53
  usage: { inToks: number; outToks: number };
43
54
  }
44
55
 
45
- /** Tree marker for one task row. */
56
+ /** Tree marker for one task row. A rolled-back task is `pending` but still
57
+ * carries the evidence of its previous attempt, so it gets its own marker
58
+ * rather than reading as untouched work. */
46
59
  export function taskMarker(task: TaskView, currentId: string | null): string {
47
60
  if (taskIsTerminal(task)) return task.status === "skipped" ? "~" : "✓";
48
61
  if (task.id === currentId) return "▸";
49
- return "·";
62
+ return task.evidence === undefined ? "·" : "↺";
50
63
  }
51
64
 
52
65
  /** Truncate to at most `max` display columns, ellipsis when clipped. */
@@ -157,16 +170,50 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
157
170
  lines.push(boxRow("│", ` ${clip(cur.files.join(", "), inner - 4)}`, " ", width));
158
171
  }
159
172
  } else if (allTasksTerminal(model.tasks)) {
160
- const auditLine = model.auditFailed.length > 0
161
- ? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
162
- : model.auditRounds !== null
163
- ? "audit complete ✓"
164
- : "all tasks terminal — audit pending";
173
+ // v0.8 mis-cue guard: `audit complete ✓` ONLY when nothing is running
174
+ // and nothing is owed. A running or owed review shows its round number
175
+ // instead — the panel must look alive, never finished-then-silent.
176
+ const owed = model.checklist.some((item) => !item.done);
177
+ const nextRound = (model.auditRounds ?? 0) + 1;
178
+ const highCount = model.findings.filter((f) => f.severity === "high").length;
179
+ const auditLine = model.reviewRunning
180
+ ? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} running — read-only reviewer verifying`
181
+ : model.auditFailed.length > 0
182
+ ? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
183
+ : highCount > 0
184
+ ? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — ${highCount} high finding(s) unresolved`
185
+ : model.auditUndeterminable.length > 0
186
+ ? `audit: ${model.auditUndeterminable.length} check(s) undeterminable — verdict unreadable`
187
+ : owed
188
+ ? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — verdict pending`
189
+ : model.auditRounds !== null
190
+ ? "audit complete ✓"
191
+ : "all tasks terminal — audit pending";
165
192
  lines.push(boxRow("│", ` ${clip(auditLine, inner - 2)}`, " ", width));
166
193
  }
167
194
  if (!narrow && model.auditFailed.length > 0) {
168
195
  lines.push(boxRow("│", ` ✗ ${clip(model.auditFailed.join(", "), inner - 3)}`, " ", width));
169
196
  }
197
+ // v0.9: unresolved findings render in both phases — the fix loop's whole
198
+ // point is that the executor sees what it owes while repairing.
199
+ const highFindings = model.findings.filter((f) => f.severity === "high");
200
+ if (!narrow && highFindings.length > 0) {
201
+ lines.push(boxRow("│", ` ⚠ ${clip(`high: ${highFindings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
202
+ } else if (!narrow && model.findings.length > 0) {
203
+ lines.push(boxRow("│", ` · ${clip(`findings: ${model.findings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
204
+ }
205
+ if (!narrow && model.auditUndeterminable.length > 0) {
206
+ lines.push(boxRow("│", ` ? ${clip(model.auditUndeterminable.join(", "), inner - 3)}`, " ", width));
207
+ }
208
+ // Rolled-back work keeps the evidence of the attempt that was rolled back.
209
+ // The tree view shows it per row; the compact panel has room for a summary
210
+ // and must show it too, otherwise the retention is invisible outside the
211
+ // expanded view. Capped at two rows to protect the fixed panel height.
212
+ for (const rolled of flattenTaskViews(model.tasks)
213
+ .filter((task) => task.status === "pending" && task.evidence !== undefined)
214
+ .slice(0, 2)) {
215
+ lines.push(boxRow("│", ` ↺ ${rolled.id} ${clip(rolled.evidence!, inner - 6)}`, " ", width));
216
+ }
170
217
  lines.push(boxRow("└", "─ Ctrl+Shift+T tree", "─", width, "┘"));
171
218
  // The title row carries its own colours; every other row is muted.
172
219
  const painted = theme ? lines.map((line, i) => (i === 0 ? line : theme.fg("muted", line))) : lines;
@@ -183,7 +230,7 @@ export function deriveDashboardModel(
183
230
  topic: string,
184
231
  tasks: TaskView[],
185
232
  checklist: CheckItem[],
186
- extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; startedAt?: string; usage?: { inToks: number; outToks: number } },
233
+ extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; findings?: Array<{ id: string; severity: string; note: string; taskIds: string[] }>; startedAt?: string; usage?: { inToks: number; outToks: number } },
187
234
  ): DashboardModel {
188
235
  return {
189
236
  topic,
@@ -193,6 +240,9 @@ export function deriveDashboardModel(
193
240
  pausedReason: extra?.pausedReason,
194
241
  auditRounds: extra?.auditRounds ?? null,
195
242
  auditFailed: extra?.auditFailed ?? [],
243
+ auditUndeterminable: extra?.auditUndeterminable ?? [],
244
+ reviewRunning: extra?.reviewRunning ?? false,
245
+ findings: extra?.findings ?? [],
196
246
  startedAt: extra?.startedAt ?? new Date().toISOString(),
197
247
  usage: extra?.usage ?? { inToks: 0, outToks: 0 },
198
248
  };
@@ -204,9 +254,14 @@ export function formatDashboardSummaryLine(model: DashboardModel): string {
204
254
  const vcDone = model.checklist.filter((item) => item.done).length;
205
255
  const cur = currentTask(model.tasks);
206
256
  const wave = cur ? ` · wave ${cur.wave}` : "";
207
- const audit = model.auditRounds !== null ? ` · audit r${model.auditRounds}` : "";
257
+ // v0.9: the review token carries the budget denominator and the unresolved
258
+ // high count — visible in BOTH phases (executing repair and verifying), so
259
+ // convergence is legible exactly while the executor is fixing.
260
+ const highs = model.findings.filter((f) => f.severity === "high").length;
261
+ const audit = model.auditRounds !== null ? ` · review r${model.auditRounds}/${REVIEW_MAX_ROUNDS}` : "";
262
+ const highToken = highs > 0 ? ` · ${highs} high` : "";
208
263
  const pause = model.paused ? " · ⏸ paused" : "";
209
- return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${pause}`;
264
+ return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${highToken}${pause}`;
210
265
  }
211
266
 
212
267
  /** Expanded tree view lines (Ctrl+Shift+T overlay). Wide layout from 96 cols. */
@@ -226,18 +281,41 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
226
281
  if (wide && task.deps.length > 0) line += ` (deps: ${task.deps.join(", ")})`;
227
282
  if (task.status === "skipped" && task.skipReason) line += ` ~${task.skipReason}`;
228
283
  if (task.status === "complete" && task.evidence) line += ` ✓${clip(task.evidence, 40)}`;
284
+ // Rolled-back work keeps its evidence, and that evidence is the only
285
+ // record of what the previous attempt did.
286
+ if (task.status === "pending" && task.evidence) line += ` ↺${clip(task.evidence, 40)}`;
229
287
  lines.push(line);
230
288
  for (const child of task.children) row(child, depth + 1);
231
- }; for (const task of model.tasks) row(task, 0);
289
+ };
290
+ for (const task of model.tasks) row(task, 0);
232
291
  lines.push("");
233
292
  lines.push("Verification checks:");
234
293
  for (const item of model.checklist) {
235
294
  const mark = model.auditFailed.includes(item.id) ? "✗" : item.done ? "☑" : "☐";
236
295
  lines.push(` ${mark} ${item.id}${wide ? ` ${clip(item.text.split(";")[1] ?? item.text, Math.min(80, width))}` : ""}`);
237
296
  }
238
- if (model.auditRounds !== null) {
297
+ if (model.findings.length > 0) {
298
+ lines.push("");
299
+ lines.push("Findings (stable ids; absence from the newest round = resolved):");
300
+ for (const f of model.findings) {
301
+ const mark = f.severity === "high" ? "⚠" : f.severity === "malformed" ? "?" : "·";
302
+ lines.push(` ${mark} ${f.id} (${f.severity}${f.taskIds.length ? `, ${f.taskIds.join(", ")}` : ""}): ${wide ? clip(f.note, 72) : clip(f.note, 40)}`);
303
+ }
304
+ }
305
+ if (model.reviewRunning) {
306
+ lines.push("");
307
+ lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${REVIEW_MAX_ROUNDS} running`);
308
+ } else if (model.auditRounds !== null) {
239
309
  lines.push("");
240
- lines.push(`Completion audit: round ${model.auditRounds}${model.auditFailed.length > 0 ? ` — failed: ${model.auditFailed.join(", ")}` : " — passed ✓"}`);
310
+ const highIds = model.findings.filter((f) => f.severity === "high").map((f) => f.id);
311
+ const verdict = model.auditFailed.length > 0
312
+ ? ` — failed: ${model.auditFailed.join(", ")}${highIds.length > 0 ? `; high: ${highIds.join(", ")}` : ""}`
313
+ : highIds.length > 0
314
+ ? ` — high findings unresolved: ${highIds.join(", ")}`
315
+ : model.auditUndeterminable.length > 0
316
+ ? ` — undeterminable: ${model.auditUndeterminable.join(", ")}`
317
+ : " — passed ✓";
318
+ lines.push(`Execution review: round ${model.auditRounds}${verdict}`);
241
319
  }
242
320
  const painted = theme ? lines.map((line) => theme.fg("muted", line)) : lines;
243
321
  return clampLines(painted, width);