pi-plans 0.7.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +92 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +14 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +306 -63
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +91 -13
- package/src/exec.ts +835 -142
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +40 -8
- package/src/resume-command.ts +19 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +62 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +93 -6
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +299 -16
- package/tests/dashboard.test.ts +202 -2
- package/tests/exec-review-loop.test.ts +724 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/refine-ui.test.ts +25 -2
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +105 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
- package/tools/refine.ts +22 -3
package/src/auditor.ts
CHANGED
|
@@ -1,87 +1,249 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
* independent read-only subagent verifies the
|
|
4
|
-
* against the worktree. Failed checks roll their
|
|
5
|
-
* pending (exclusively inside the
|
|
6
|
-
*
|
|
7
|
-
*
|
|
2
|
+
* Execution reviewer (v0.8): when every task
|
|
3
|
+
* reaches a terminal state, an independent read-only subagent verifies the
|
|
4
|
+
* plan's verification checks against the worktree. Failed checks roll their
|
|
5
|
+
* covered tasks back to pending (exclusively inside the review flow). The
|
|
6
|
+
* loop is bounded at REVIEW_MAX_ROUNDS committed rounds; exhaustion pauses
|
|
7
|
+
* the run for the user in every mode (fail-closed, never a hang and never a
|
|
8
|
+
* silent stop) — only an explicit /plans-execute confirmation grants a fresh
|
|
9
|
+
* budget.
|
|
8
10
|
*/
|
|
9
11
|
|
|
10
12
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
11
13
|
import * as fs from "node:fs";
|
|
12
|
-
import
|
|
13
|
-
import type
|
|
14
|
+
import * as path from "node:path";
|
|
15
|
+
import { auditableChecks, flattenTaskViews, skippedPassCheckIds, type TaskView } from "./tasks.ts";
|
|
16
|
+
import { normalizeTaskId, type CheckItem } from "./plan.ts";
|
|
14
17
|
import { messaging } from "./messaging.ts";
|
|
15
18
|
|
|
16
|
-
|
|
19
|
+
/** Budget cap: committed rounds per user-granted budget. Discarded (fingerprint-changed) attempts do not count. */
|
|
20
|
+
export const REVIEW_MAX_ROUNDS = 5;
|
|
21
|
+
|
|
22
|
+
/** Legacy alias for one release: checkpoints and old builds still know this name. */
|
|
23
|
+
export const AUDIT_MAX_ROUNDS = REVIEW_MAX_ROUNDS;
|
|
17
24
|
|
|
18
25
|
export interface AuditOutcome {
|
|
19
26
|
round: number;
|
|
20
27
|
passed: string[];
|
|
21
28
|
failed: string[];
|
|
22
|
-
/**
|
|
23
|
-
|
|
29
|
+
/** Checks whose report carried no readable verdict, or whose evidence was
|
|
30
|
+
* inconclusive. Never treated as `failed`: see src/exec.ts. */
|
|
31
|
+
undeterminable: string[];
|
|
32
|
+
/** Severity-graded implementation findings from this round (v0.9). Absent
|
|
33
|
+
* on legacy shapes means "no findings reported"; the parse boundary always
|
|
34
|
+
* sets a concrete array so hand-written construction sites cannot silently
|
|
35
|
+
* drop them. */
|
|
36
|
+
findings?: ReviewFinding[];
|
|
24
37
|
report: string;
|
|
25
38
|
}
|
|
26
39
|
|
|
40
|
+
/** Severity vocabulary for implementation findings; mirrors the plan
|
|
41
|
+
* priority words. `malformed` marks a bullet the grammar parser could not
|
|
42
|
+
* read — it is recorded and displayed but never drives a rollback. */
|
|
43
|
+
export type FindingSeverity = "high" | "medium" | "low" | "malformed";
|
|
44
|
+
|
|
45
|
+
export interface ReviewFinding {
|
|
46
|
+
/** Stable id, normalized uppercase (`F-001`). Reused verbatim across
|
|
47
|
+
* rounds while the problem persists; absence from the newest round's
|
|
48
|
+
* report is the resolution signal. */
|
|
49
|
+
id: string;
|
|
50
|
+
severity: FindingSeverity;
|
|
51
|
+
/** Task ids owning the defect (normalized), `[]` when unmapped. */
|
|
52
|
+
taskIds: string[];
|
|
53
|
+
/** Required for unmapped high findings: the runner appends this as a new
|
|
54
|
+
* plan task mechanically, so it must be a self-contained imperative title. */
|
|
55
|
+
proposedTask?: string;
|
|
56
|
+
note: string;
|
|
57
|
+
evidence: string;
|
|
58
|
+
/** Original bullet, for degraded records and round reports. */
|
|
59
|
+
raw: string;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Parse the verdict lines of an audit report against the checks that are
|
|
63
|
+
* still pending. A pending check with no readable verdict is
|
|
64
|
+
* `undeterminable`, never `failed` — the caller must be able to tell "the
|
|
65
|
+
* work is wrong" apart from "I could not read the answer". */
|
|
66
|
+
export interface ParsedAudit {
|
|
67
|
+
passed: string[];
|
|
68
|
+
failed: string[];
|
|
69
|
+
undeterminable: string[];
|
|
70
|
+
findings: ReviewFinding[];
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** One finding bullet's field-capture helper: lazily up to the next
|
|
74
|
+
* `; <known-field>:` boundary or the end of the line, so free text in one
|
|
75
|
+
* field cannot swallow the next. */
|
|
76
|
+
function captureField(line: string, field: string): string | undefined {
|
|
77
|
+
const m = line.match(new RegExp(`${field}:\\s*(.*?)(?=;\\s*(?:severity|tasks|proposed-task|note|evidence):|$)`, "i"));
|
|
78
|
+
return m ? m[1].trim().replace(/^[*_`~]+|[*_`~]+$/g, "") : undefined;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Parse the findings bullets of a review report. Degrade, never crash, never
|
|
82
|
+
* roll back: a bullet with an F-### id but unreadable fields is recorded with
|
|
83
|
+
* severity "malformed" (visible, non-blocking), exactly like the verdict
|
|
84
|
+
* parser's emphasis tolerance — a strict grammar once silently discarded
|
|
85
|
+
* verdicts and fail-closed rolled correct work back. */
|
|
86
|
+
export function parseFindings(report: string, knownTaskIds?: Set<string>): ReviewFinding[] {
|
|
87
|
+
const findings: ReviewFinding[] = [];
|
|
88
|
+
const seen = new Set<string>();
|
|
89
|
+
for (const match of report.matchAll(/^\s*[-*]\s+`?(F-\d+)`?\b/gim)) {
|
|
90
|
+
const id = match[1].toUpperCase();
|
|
91
|
+
if (seen.has(id)) continue; // conflicting duplicates resolve to the first
|
|
92
|
+
seen.add(id);
|
|
93
|
+
// The bullet regex's leading \s* may swallow the preceding newline, so
|
|
94
|
+
// slice(index) can start mid-whitespace; strip it before taking the line.
|
|
95
|
+
const line = match.input!.slice(match.index!).replace(/^[\s]+/, "").split(/\n/)[0];
|
|
96
|
+
const severityRaw = captureField(line, "severity")?.toLowerCase();
|
|
97
|
+
const tasksRaw = captureField(line, "tasks");
|
|
98
|
+
const taskIds = (tasksRaw ?? "")
|
|
99
|
+
.split(/[,,,、]/)
|
|
100
|
+
.map((token) => normalizeTaskId(token.trim()))
|
|
101
|
+
.filter((t): t is string => t !== null)
|
|
102
|
+
.filter((t) => !knownTaskIds || knownTaskIds.has(t));
|
|
103
|
+
const severity: FindingSeverity =
|
|
104
|
+
severityRaw === "high" || severityRaw === "medium" || severityRaw === "low" ? severityRaw : "malformed";
|
|
105
|
+
if (severity === "malformed" || !tasksRaw) {
|
|
106
|
+
// Unreadable severity, or the grammar's mandatory tasks field missing:
|
|
107
|
+
// degrade to a recorded non-blocking entry.
|
|
108
|
+
findings.push({ id, severity: "malformed", taskIds: [], note: captureField(line, "note") ?? line.trim(), evidence: captureField(line, "evidence") ?? "", raw: line.trim() });
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
findings.push({
|
|
112
|
+
id,
|
|
113
|
+
severity,
|
|
114
|
+
taskIds,
|
|
115
|
+
proposedTask: captureField(line, "proposed-task") || undefined,
|
|
116
|
+
note: captureField(line, "note") ?? "",
|
|
117
|
+
evidence: captureField(line, "evidence") ?? "",
|
|
118
|
+
raw: line.trim(),
|
|
119
|
+
});
|
|
120
|
+
}
|
|
121
|
+
return findings;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Checks this round must judge: auditable (covering at least one task) and
|
|
125
|
+
* not already done. The brief and the coverage self-check both use this, so a
|
|
126
|
+
* well-formed round-2 report that omits an already-done check is not mistaken
|
|
127
|
+
* for a contract violation. */
|
|
128
|
+
export function auditablePendingChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
|
|
129
|
+
return auditableChecks(checklist, tasks).filter((item) => !item.done);
|
|
130
|
+
}
|
|
131
|
+
|
|
27
132
|
/** Build the audit brief for the read-only subagent. Exported for tests. */
|
|
28
|
-
export function buildAuditTask(
|
|
29
|
-
|
|
133
|
+
export function buildAuditTask(
|
|
134
|
+
planPath: string,
|
|
135
|
+
checklist: CheckItem[],
|
|
136
|
+
tasks: TaskView[],
|
|
137
|
+
round: number,
|
|
138
|
+
priorFindings: ReviewFinding[] = [],
|
|
139
|
+
): string {
|
|
140
|
+
const checks = auditablePendingChecks(checklist, tasks)
|
|
30
141
|
.map((check) => `- \`${check.id}\`: ${check.text}`)
|
|
31
142
|
.join("\n");
|
|
32
|
-
|
|
143
|
+
const taskList = flattenTaskViews(tasks)
|
|
144
|
+
.map((t) => `- \`${t.id}\`: ${t.title} (${t.status})`)
|
|
145
|
+
.join("\n");
|
|
146
|
+
const prior = priorFindings.length
|
|
147
|
+
? `Unresolved findings from earlier rounds (reuse these exact ids while the problem persists; a problem is resolved only by no longer reporting it):
|
|
148
|
+
|
|
149
|
+
${priorFindings.map((f) => `- \`${f.id}\` — severity: ${f.severity}; tasks: ${f.taskIds.join(", ") || "none"}; note: ${f.note}`).join("\n")}`
|
|
150
|
+
: "(none — this is the first round with findings in scope)";
|
|
151
|
+
return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks, and report implementation findings that drive the fix loop.
|
|
33
152
|
|
|
34
|
-
Target plan: ${planPath} (
|
|
153
|
+
Target plan: ${planPath} (review round ${round})
|
|
35
154
|
|
|
36
155
|
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
|
|
37
156
|
|
|
38
157
|
Evidence: inspect the repository with read, grep, find, ls, and targeted commands (bash is not granted — rely on the read tools) before judging each check. Tests may be referenced from their recorded evidence; do not re-run them.
|
|
39
158
|
|
|
40
|
-
Checks to verify (only these; checks covering no task are excluded):
|
|
159
|
+
Checks to verify (only these; checks covering no task and checks already satisfied in an earlier round are excluded):
|
|
41
160
|
|
|
42
161
|
${checks}
|
|
43
162
|
|
|
44
|
-
|
|
163
|
+
Plan tasks (the only ids valid in a finding's tasks field):
|
|
164
|
+
|
|
165
|
+
${taskList}
|
|
166
|
+
|
|
167
|
+
${prior}
|
|
45
168
|
|
|
46
|
-
|
|
169
|
+
Output: exactly two sections, in this order.
|
|
47
170
|
|
|
48
|
-
|
|
171
|
+
1. Verification verdicts — one section per check, in the order above:
|
|
172
|
+
|
|
173
|
+
- \`VC-###\` — verdict: pass | fail | undeterminable; evidence: <repo path/command proving it>; note: <one line>.
|
|
174
|
+
|
|
175
|
+
Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence.
|
|
176
|
+
|
|
177
|
+
2. Implementation findings — one bullet per defect anywhere in the implemented change (not only what the checks cover), exact line grammar:
|
|
178
|
+
|
|
179
|
+
- \`F-###\` — severity: high | medium | low; tasks: Task-N, Task-M | none; proposed-task: <imperative one-line title>; note: <one line>; evidence: <repo path or quoted excerpt>
|
|
180
|
+
|
|
181
|
+
\`high\` wakes the executor for a fix round; \`medium\`/\`low\` are recorded. When no existing task owns the defect use \`tasks: none\` and (required for high) a self-contained \`proposed-task:\` title. If nothing is worth reporting, emit exactly \`- none.\` under the findings heading.`;
|
|
49
182
|
}
|
|
50
183
|
|
|
51
184
|
/** Parse the audit subagent's verdict lines. Exported for tests. */
|
|
52
|
-
export function parseAuditReport(report: string,
|
|
185
|
+
export function parseAuditReport(report: string, pendingVcIds: string[], knownTaskIds?: Set<string>): ParsedAudit {
|
|
53
186
|
const passed: string[] = [];
|
|
54
187
|
const failed: string[] = [];
|
|
55
|
-
const known = new Set(
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
188
|
+
const known = new Set(pendingVcIds.map((id) => id.toUpperCase()));
|
|
189
|
+
// The reviewer writes Markdown, so a verdict may carry emphasis
|
|
190
|
+
// (`**pass**`, `*pass*`, `_pass_`, `` `pass` ``). Requiring a bare token
|
|
191
|
+
// silently discarded such verdicts, and the caller's fail-closed rule then
|
|
192
|
+
// marked every check failed and rolled the whole run back -- reporting
|
|
193
|
+
// correct work as failure. Tolerate the markers; \b keeps `passed` and
|
|
194
|
+
// `passing` from matching.
|
|
195
|
+
// v0.9: finding bullets (F-###) are excluded from verdict scanning — a
|
|
196
|
+
// finding's note may cite a VC id, and that must never register a verdict.
|
|
197
|
+
// v0.9.1 (F-011): the contract promises "one section per check" and an
|
|
198
|
+
// equally literal reading puts the id in a `### VC-###` heading with the
|
|
199
|
+
// verdict on a line of its own below it. The old same-line-only scan
|
|
200
|
+
// parsed such reports to ZERO verdicts and burned whole budgets as
|
|
201
|
+
// undeterminable. The scan is now section-aware: a line that NAMES a
|
|
202
|
+
// known check at a heading/bullet start opens that check's section, and a
|
|
203
|
+
// bare `verdict:` token attributes to the nearest open section; an
|
|
204
|
+
// id-and-verdict pair on one line stays direct.
|
|
205
|
+
const record = (id: string, verdict: string): void => {
|
|
206
|
+
if (verdict === "pass") passed.push(id);
|
|
207
|
+
else if (verdict === "fail") failed.push(id);
|
|
208
|
+
};
|
|
209
|
+
const verdictToken = /verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i;
|
|
210
|
+
let current: string | null = null;
|
|
211
|
+
for (const line of report.split(/\n/)) {
|
|
212
|
+
if (/^\s*[-*]\s+`?F-\d+`?\b/i.test(line)) continue; // finding bullet
|
|
213
|
+
const sectionId = line.match(/^\s*(?:#{1,6}\s+|[-*]\s+)?`?(VC-\d+)`?\b/i);
|
|
214
|
+
if (sectionId) {
|
|
215
|
+
const id = sectionId[1].toUpperCase();
|
|
216
|
+
if (known.has(id)) current = id;
|
|
217
|
+
}
|
|
218
|
+
const direct = line.match(/`?(VC-\d+)`?[^\n]*?verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i);
|
|
219
|
+
if (direct) {
|
|
220
|
+
const id = direct[1].toUpperCase();
|
|
221
|
+
if (known.has(id)) record(id, direct[2].toLowerCase());
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
const bare = line.match(verdictToken);
|
|
225
|
+
if (bare && current) record(current, bare[1].toLowerCase());
|
|
60
226
|
}
|
|
61
|
-
//
|
|
62
|
-
for (const id of [...passed]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
|
|
63
|
-
|
|
227
|
+
// A check with conflicting verdicts resolves to fail: the reader saw both.
|
|
228
|
+
for (const id of [...new Set(passed)]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
|
|
229
|
+
const undecided = new Set(known);
|
|
230
|
+
for (const id of [...passed, ...failed]) undecided.delete(id);
|
|
231
|
+
return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided], findings: parseFindings(report, knownTaskIds) };
|
|
64
232
|
}
|
|
65
233
|
|
|
66
|
-
/** Pure decision core:
|
|
67
|
-
*
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
}
|
|
80
|
-
const rolledBack: string[] = [];
|
|
81
|
-
for (const id of failed) {
|
|
82
|
-
rolledBack.push(...auditRollbackSet(tasks, checklist, id));
|
|
83
|
-
}
|
|
84
|
-
return { round, passed, failed, rolledBack, report };
|
|
234
|
+
/** Pure decision core: classify a parsed report into the audit outcome. It
|
|
235
|
+
* touches neither the checklist nor the task tree — the caller owns writing
|
|
236
|
+
* `done` and computing the rollback set, so there is exactly one authority for
|
|
237
|
+
* both. Exported for tests. */
|
|
238
|
+
export function applyAuditOutcome(round: number, parsed: ParsedAudit, report: string): AuditOutcome {
|
|
239
|
+
return {
|
|
240
|
+
round,
|
|
241
|
+
passed: parsed.passed,
|
|
242
|
+
failed: parsed.failed,
|
|
243
|
+
undeterminable: parsed.undeterminable,
|
|
244
|
+
findings: parsed.findings,
|
|
245
|
+
report,
|
|
246
|
+
};
|
|
85
247
|
}
|
|
86
248
|
|
|
87
249
|
/** Skipped-pass checks (all covered tasks skipped) pass without audit. */
|
|
@@ -89,9 +251,13 @@ export function presolvedCheckIds(checklist: CheckItem[], tasks: TaskView[]): st
|
|
|
89
251
|
return skippedPassCheckIds(checklist, tasks);
|
|
90
252
|
}
|
|
91
253
|
|
|
92
|
-
/** Spawn the audit subagent and
|
|
93
|
-
*
|
|
94
|
-
*
|
|
254
|
+
/** Spawn the audit subagent and classify its report. Returns `{ cancelled: true }`
|
|
255
|
+
* when the child was aborted (session shutdown, stop, restore, tree switch) —
|
|
256
|
+
* such a round burns no budget and sends no wake; `null` when the subagent
|
|
257
|
+
* itself failed to run (treated as an all-undeterminable committed round; the
|
|
258
|
+
* caller counts it against the round cap). */
|
|
259
|
+
export type AuditRoundResult = AuditOutcome | { cancelled: true } | null;
|
|
260
|
+
|
|
95
261
|
export async function runCompletionAudit(
|
|
96
262
|
ctx: ExtensionContext,
|
|
97
263
|
opts: {
|
|
@@ -99,28 +265,105 @@ export async function runCompletionAudit(
|
|
|
99
265
|
checklist: CheckItem[];
|
|
100
266
|
tasks: TaskView[];
|
|
101
267
|
round: number;
|
|
268
|
+
/** Unresolved findings from earlier committed rounds, injected into
|
|
269
|
+
* the brief so the reviewer reuses stable ids (v0.9). */
|
|
270
|
+
priorFindings?: ReviewFinding[];
|
|
102
271
|
model?: string;
|
|
272
|
+
thinkingLevel?: string;
|
|
273
|
+
timeoutMs?: number;
|
|
103
274
|
signal?: AbortSignal;
|
|
275
|
+
onProgress?: (event: import("./subagent.ts").SubagentProgressEvent) => void;
|
|
104
276
|
},
|
|
105
|
-
): Promise<
|
|
277
|
+
): Promise<AuditRoundResult> {
|
|
106
278
|
const { runPiSubagent } = await import("./subagent.ts");
|
|
107
|
-
const
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
279
|
+
const pending = auditablePendingChecks(opts.checklist, opts.tasks);
|
|
280
|
+
const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round, opts.priorFindings ?? []);
|
|
281
|
+
// agents/auditor.md, not agents/reviewer.md: the reviewer prompt mandates a
|
|
282
|
+
// plan-review shape (`## Findings` / `## Questions`, F-###) and never says
|
|
283
|
+
// "verdict", so auditing under it produced reports this parser could not
|
|
284
|
+
// read at all -- every check then fell through to fail-closed.
|
|
285
|
+
// v0.8: a missing agent definition is a hard error — the silent inline
|
|
286
|
+
// fallback once swapped in a minimal prompt that produced unparsable
|
|
287
|
+
// reports and burned whole audit budgets as undeterminable.
|
|
288
|
+
const agentPrompt = fs.readFileSync(new URL("../agents/execution-reviewer.md", import.meta.url), "utf8");
|
|
114
289
|
const result = await runPiSubagent({
|
|
115
|
-
systemPrompt:
|
|
290
|
+
systemPrompt: agentPrompt,
|
|
116
291
|
task,
|
|
117
292
|
cwd: ctx.cwd,
|
|
118
293
|
model: opts.model,
|
|
294
|
+
thinkingLevel: opts.thinkingLevel,
|
|
295
|
+
timeoutMs: opts.timeoutMs,
|
|
119
296
|
tools: ["read", "grep", "find", "ls"],
|
|
120
297
|
signal: opts.signal,
|
|
298
|
+
onProgress: opts.onProgress,
|
|
121
299
|
});
|
|
122
|
-
if (!result.ok)
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
300
|
+
if (!result.ok) {
|
|
301
|
+
if (result.cancelled === true) return { cancelled: true };
|
|
302
|
+
return null;
|
|
303
|
+
}
|
|
304
|
+
const parsed = parseAuditReport(result.output, pending.map((item) => item.id), new Set(flattenTaskViews(opts.tasks).map((t) => t.id)));
|
|
305
|
+
messaging().appendEntry("pi-plans-audit", {
|
|
306
|
+
planPath: opts.planPath,
|
|
307
|
+
round: opts.round,
|
|
308
|
+
passed: parsed.passed,
|
|
309
|
+
failed: parsed.failed,
|
|
310
|
+
undeterminable: parsed.undeterminable,
|
|
311
|
+
highFindings: parsed.findings.filter((f) => f.severity === "high").map((f) => f.id),
|
|
312
|
+
});
|
|
313
|
+
return applyAuditOutcome(opts.round, parsed, result.output);
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
/** Persist one review-round attempt under `<run-dir>/execution-review/`.
|
|
317
|
+
* The attempt index (not the budget round) names the file, so a discarded
|
|
318
|
+
* attempt's evidence survives its re-run: round-<budgetRound>-attempt-<k>.md.
|
|
319
|
+
* These files are the only cross-session record of what a round saw — the
|
|
320
|
+
* in-flight marker itself is memory-only. Best-effort: an unwritable run dir
|
|
321
|
+
* must never fail the loop itself. */
|
|
322
|
+
export function writeReviewRoundReport(
|
|
323
|
+
runDir: string,
|
|
324
|
+
entry: {
|
|
325
|
+
budgetRound: number;
|
|
326
|
+
attempt: number;
|
|
327
|
+
outcome: "passed" | "failed" | "undeterminable" | "discarded" | "spawn-failed" | "cancelled";
|
|
328
|
+
passed: string[];
|
|
329
|
+
failed: string[];
|
|
330
|
+
undeterminable: string[];
|
|
331
|
+
/** v0.9 findings from this round; recorded in the round report so the
|
|
332
|
+
* file stays the full evidence record. */
|
|
333
|
+
findings?: ReviewFinding[];
|
|
334
|
+
discardedReason?: string;
|
|
335
|
+
fingerprintCaptured?: string;
|
|
336
|
+
fingerprintFound?: string;
|
|
337
|
+
coveredTaskIds: string[];
|
|
338
|
+
report: string;
|
|
339
|
+
},
|
|
340
|
+
): string | null {
|
|
341
|
+
try {
|
|
342
|
+
const dir = path.join(runDir, "execution-review");
|
|
343
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
344
|
+
const file = path.join(dir, `round-${entry.budgetRound}-attempt-${entry.attempt}.md`);
|
|
345
|
+
const lines = [
|
|
346
|
+
`# Execution review round ${entry.budgetRound} (attempt ${entry.attempt})`,
|
|
347
|
+
"",
|
|
348
|
+
`- outcome: ${entry.outcome}`,
|
|
349
|
+
`- passed: ${entry.passed.join(", ") || "(none)"}`,
|
|
350
|
+
`- failed: ${entry.failed.join(", ") || "(none)"}`,
|
|
351
|
+
`- undeterminable: ${entry.undeterminable.join(", ") || "(none)"}`,
|
|
352
|
+
`- high findings: ${entry.findings?.filter((f) => f.severity === "high").map((f) => f.id).join(", ") || "(none)"}`,
|
|
353
|
+
...(entry.findings?.length ? [`- findings: ${entry.findings.map((f) => `${f.id} (${f.severity}${f.proposedTask ? "; proposed: " + f.proposedTask : ""})`).join(" | ")}`] : []),
|
|
354
|
+
...(entry.discardedReason ? [`- discarded: ${entry.discardedReason}`] : []),
|
|
355
|
+
`- fingerprint (captured): ${entry.fingerprintCaptured ?? "(n/a)"}`,
|
|
356
|
+
`- fingerprint (at resolve): ${entry.fingerprintFound ?? "(n/a)"}`,
|
|
357
|
+
`- covered tasks: ${entry.coveredTaskIds.join(", ") || "(none)"}`,
|
|
358
|
+
"",
|
|
359
|
+
"## Report",
|
|
360
|
+
"",
|
|
361
|
+
entry.report,
|
|
362
|
+
"",
|
|
363
|
+
];
|
|
364
|
+
fs.writeFileSync(file, lines.join("\n"), "utf8");
|
|
365
|
+
return file;
|
|
366
|
+
} catch {
|
|
367
|
+
return null;
|
|
368
|
+
}
|
|
126
369
|
}
|
|
@@ -284,7 +284,12 @@ export async function applyGraphCommand(args: string, ctx: CommandContext): Prom
|
|
|
284
284
|
errors > 0 ? "error" : "info",
|
|
285
285
|
);
|
|
286
286
|
const active = resolveActiveRun(ctx.sessionManager, ctx.cwd);
|
|
287
|
-
if (active)
|
|
287
|
+
if (active) {
|
|
288
|
+
// Never demote a verifying run back to executing: the execution-review
|
|
289
|
+
// loop owns the status until it converges (or pauses at the round cap).
|
|
290
|
+
const current = getRun(ctx.cwd, active.run_id)?.status;
|
|
291
|
+
if (current !== "verifying") setRunStatus(ctx.cwd, active.run_id, "executing");
|
|
292
|
+
}
|
|
288
293
|
}
|
|
289
294
|
|
|
290
295
|
export async function graphStatusCommand(_args: string, ctx: CommandContext): Promise<void> {
|
package/src/dashboard.ts
CHANGED
|
@@ -16,6 +16,7 @@
|
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import type { CheckItem } from "./plan.ts";
|
|
19
|
+
import { REVIEW_MAX_ROUNDS } from "./auditor.ts";
|
|
19
20
|
import { truncateToWidth, visibleWidth } from "./refine-ui-helpers.ts";
|
|
20
21
|
import {
|
|
21
22
|
allTasksTerminal,
|
|
@@ -38,15 +39,27 @@ export interface DashboardModel {
|
|
|
38
39
|
/** Audit round counter; null = audit not yet started. */
|
|
39
40
|
auditRounds: number | null;
|
|
40
41
|
auditFailed: string[];
|
|
42
|
+
/** Checks whose verdict the auditor could not be read for. Neither passed
|
|
43
|
+
* nor failed: shown so the run does not read as finished. */
|
|
44
|
+
auditUndeterminable: string[];
|
|
45
|
+
/** v0.8: true while an execution-review round is in flight. The panel must
|
|
46
|
+
* never render `audit complete ✓` while this is set — the round-1 mis-cue
|
|
47
|
+
* showed the tick for the whole duration of a running audit. */
|
|
48
|
+
reviewRunning: boolean;
|
|
49
|
+
/** v0.9: unresolved findings from the newest committed review round
|
|
50
|
+
* (stable ids). High entries block completion; the rest are recorded. */
|
|
51
|
+
findings: Array<{ id: string; severity: string; note: string; taskIds: string[] }>;
|
|
41
52
|
startedAt: string;
|
|
42
53
|
usage: { inToks: number; outToks: number };
|
|
43
54
|
}
|
|
44
55
|
|
|
45
|
-
/** Tree marker for one task row.
|
|
56
|
+
/** Tree marker for one task row. A rolled-back task is `pending` but still
|
|
57
|
+
* carries the evidence of its previous attempt, so it gets its own marker
|
|
58
|
+
* rather than reading as untouched work. */
|
|
46
59
|
export function taskMarker(task: TaskView, currentId: string | null): string {
|
|
47
60
|
if (taskIsTerminal(task)) return task.status === "skipped" ? "~" : "✓";
|
|
48
61
|
if (task.id === currentId) return "▸";
|
|
49
|
-
return "·";
|
|
62
|
+
return task.evidence === undefined ? "·" : "↺";
|
|
50
63
|
}
|
|
51
64
|
|
|
52
65
|
/** Truncate to at most `max` display columns, ellipsis when clipped. */
|
|
@@ -157,16 +170,50 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
|
|
|
157
170
|
lines.push(boxRow("│", ` ${clip(cur.files.join(", "), inner - 4)}`, " ", width));
|
|
158
171
|
}
|
|
159
172
|
} else if (allTasksTerminal(model.tasks)) {
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
173
|
+
// v0.8 mis-cue guard: `audit complete ✓` ONLY when nothing is running
|
|
174
|
+
// and nothing is owed. A running or owed review shows its round number
|
|
175
|
+
// instead — the panel must look alive, never finished-then-silent.
|
|
176
|
+
const owed = model.checklist.some((item) => !item.done);
|
|
177
|
+
const nextRound = (model.auditRounds ?? 0) + 1;
|
|
178
|
+
const highCount = model.findings.filter((f) => f.severity === "high").length;
|
|
179
|
+
const auditLine = model.reviewRunning
|
|
180
|
+
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} running — read-only reviewer verifying`
|
|
181
|
+
: model.auditFailed.length > 0
|
|
182
|
+
? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
|
|
183
|
+
: highCount > 0
|
|
184
|
+
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — ${highCount} high finding(s) unresolved`
|
|
185
|
+
: model.auditUndeterminable.length > 0
|
|
186
|
+
? `audit: ${model.auditUndeterminable.length} check(s) undeterminable — verdict unreadable`
|
|
187
|
+
: owed
|
|
188
|
+
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — verdict pending`
|
|
189
|
+
: model.auditRounds !== null
|
|
190
|
+
? "audit complete ✓"
|
|
191
|
+
: "all tasks terminal — audit pending";
|
|
165
192
|
lines.push(boxRow("│", ` ${clip(auditLine, inner - 2)}`, " ", width));
|
|
166
193
|
}
|
|
167
194
|
if (!narrow && model.auditFailed.length > 0) {
|
|
168
195
|
lines.push(boxRow("│", ` ✗ ${clip(model.auditFailed.join(", "), inner - 3)}`, " ", width));
|
|
169
196
|
}
|
|
197
|
+
// v0.9: unresolved findings render in both phases — the fix loop's whole
|
|
198
|
+
// point is that the executor sees what it owes while repairing.
|
|
199
|
+
const highFindings = model.findings.filter((f) => f.severity === "high");
|
|
200
|
+
if (!narrow && highFindings.length > 0) {
|
|
201
|
+
lines.push(boxRow("│", ` ⚠ ${clip(`high: ${highFindings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
|
|
202
|
+
} else if (!narrow && model.findings.length > 0) {
|
|
203
|
+
lines.push(boxRow("│", ` · ${clip(`findings: ${model.findings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
|
|
204
|
+
}
|
|
205
|
+
if (!narrow && model.auditUndeterminable.length > 0) {
|
|
206
|
+
lines.push(boxRow("│", ` ? ${clip(model.auditUndeterminable.join(", "), inner - 3)}`, " ", width));
|
|
207
|
+
}
|
|
208
|
+
// Rolled-back work keeps the evidence of the attempt that was rolled back.
|
|
209
|
+
// The tree view shows it per row; the compact panel has room for a summary
|
|
210
|
+
// and must show it too, otherwise the retention is invisible outside the
|
|
211
|
+
// expanded view. Capped at two rows to protect the fixed panel height.
|
|
212
|
+
for (const rolled of flattenTaskViews(model.tasks)
|
|
213
|
+
.filter((task) => task.status === "pending" && task.evidence !== undefined)
|
|
214
|
+
.slice(0, 2)) {
|
|
215
|
+
lines.push(boxRow("│", ` ↺ ${rolled.id} ${clip(rolled.evidence!, inner - 6)}`, " ", width));
|
|
216
|
+
}
|
|
170
217
|
lines.push(boxRow("└", "─ Ctrl+Shift+T tree", "─", width, "┘"));
|
|
171
218
|
// The title row carries its own colours; every other row is muted.
|
|
172
219
|
const painted = theme ? lines.map((line, i) => (i === 0 ? line : theme.fg("muted", line))) : lines;
|
|
@@ -183,7 +230,7 @@ export function deriveDashboardModel(
|
|
|
183
230
|
topic: string,
|
|
184
231
|
tasks: TaskView[],
|
|
185
232
|
checklist: CheckItem[],
|
|
186
|
-
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
233
|
+
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; findings?: Array<{ id: string; severity: string; note: string; taskIds: string[] }>; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
187
234
|
): DashboardModel {
|
|
188
235
|
return {
|
|
189
236
|
topic,
|
|
@@ -193,6 +240,9 @@ export function deriveDashboardModel(
|
|
|
193
240
|
pausedReason: extra?.pausedReason,
|
|
194
241
|
auditRounds: extra?.auditRounds ?? null,
|
|
195
242
|
auditFailed: extra?.auditFailed ?? [],
|
|
243
|
+
auditUndeterminable: extra?.auditUndeterminable ?? [],
|
|
244
|
+
reviewRunning: extra?.reviewRunning ?? false,
|
|
245
|
+
findings: extra?.findings ?? [],
|
|
196
246
|
startedAt: extra?.startedAt ?? new Date().toISOString(),
|
|
197
247
|
usage: extra?.usage ?? { inToks: 0, outToks: 0 },
|
|
198
248
|
};
|
|
@@ -204,9 +254,14 @@ export function formatDashboardSummaryLine(model: DashboardModel): string {
|
|
|
204
254
|
const vcDone = model.checklist.filter((item) => item.done).length;
|
|
205
255
|
const cur = currentTask(model.tasks);
|
|
206
256
|
const wave = cur ? ` · wave ${cur.wave}` : "";
|
|
207
|
-
|
|
257
|
+
// v0.9: the review token carries the budget denominator and the unresolved
|
|
258
|
+
// high count — visible in BOTH phases (executing repair and verifying), so
|
|
259
|
+
// convergence is legible exactly while the executor is fixing.
|
|
260
|
+
const highs = model.findings.filter((f) => f.severity === "high").length;
|
|
261
|
+
const audit = model.auditRounds !== null ? ` · review r${model.auditRounds}/${REVIEW_MAX_ROUNDS}` : "";
|
|
262
|
+
const highToken = highs > 0 ? ` · ${highs} high` : "";
|
|
208
263
|
const pause = model.paused ? " · ⏸ paused" : "";
|
|
209
|
-
return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${pause}`;
|
|
264
|
+
return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${highToken}${pause}`;
|
|
210
265
|
}
|
|
211
266
|
|
|
212
267
|
/** Expanded tree view lines (Ctrl+Shift+T overlay). Wide layout from 96 cols. */
|
|
@@ -226,18 +281,41 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
|
|
|
226
281
|
if (wide && task.deps.length > 0) line += ` (deps: ${task.deps.join(", ")})`;
|
|
227
282
|
if (task.status === "skipped" && task.skipReason) line += ` ~${task.skipReason}`;
|
|
228
283
|
if (task.status === "complete" && task.evidence) line += ` ✓${clip(task.evidence, 40)}`;
|
|
284
|
+
// Rolled-back work keeps its evidence, and that evidence is the only
|
|
285
|
+
// record of what the previous attempt did.
|
|
286
|
+
if (task.status === "pending" && task.evidence) line += ` ↺${clip(task.evidence, 40)}`;
|
|
229
287
|
lines.push(line);
|
|
230
288
|
for (const child of task.children) row(child, depth + 1);
|
|
231
|
-
};
|
|
289
|
+
};
|
|
290
|
+
for (const task of model.tasks) row(task, 0);
|
|
232
291
|
lines.push("");
|
|
233
292
|
lines.push("Verification checks:");
|
|
234
293
|
for (const item of model.checklist) {
|
|
235
294
|
const mark = model.auditFailed.includes(item.id) ? "✗" : item.done ? "☑" : "☐";
|
|
236
295
|
lines.push(` ${mark} ${item.id}${wide ? ` ${clip(item.text.split(";")[1] ?? item.text, Math.min(80, width))}` : ""}`);
|
|
237
296
|
}
|
|
238
|
-
if (model.
|
|
297
|
+
if (model.findings.length > 0) {
|
|
298
|
+
lines.push("");
|
|
299
|
+
lines.push("Findings (stable ids; absence from the newest round = resolved):");
|
|
300
|
+
for (const f of model.findings) {
|
|
301
|
+
const mark = f.severity === "high" ? "⚠" : f.severity === "malformed" ? "?" : "·";
|
|
302
|
+
lines.push(` ${mark} ${f.id} (${f.severity}${f.taskIds.length ? `, ${f.taskIds.join(", ")}` : ""}): ${wide ? clip(f.note, 72) : clip(f.note, 40)}`);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
if (model.reviewRunning) {
|
|
306
|
+
lines.push("");
|
|
307
|
+
lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${REVIEW_MAX_ROUNDS} running`);
|
|
308
|
+
} else if (model.auditRounds !== null) {
|
|
239
309
|
lines.push("");
|
|
240
|
-
|
|
310
|
+
const highIds = model.findings.filter((f) => f.severity === "high").map((f) => f.id);
|
|
311
|
+
const verdict = model.auditFailed.length > 0
|
|
312
|
+
? ` — failed: ${model.auditFailed.join(", ")}${highIds.length > 0 ? `; high: ${highIds.join(", ")}` : ""}`
|
|
313
|
+
: highIds.length > 0
|
|
314
|
+
? ` — high findings unresolved: ${highIds.join(", ")}`
|
|
315
|
+
: model.auditUndeterminable.length > 0
|
|
316
|
+
? ` — undeterminable: ${model.auditUndeterminable.join(", ")}`
|
|
317
|
+
: " — passed ✓";
|
|
318
|
+
lines.push(`Execution review: round ${model.auditRounds}${verdict}`);
|
|
241
319
|
}
|
|
242
320
|
const painted = theme ? lines.map((line) => theme.fg("muted", line)) : lines;
|
|
243
321
|
return clampLines(painted, width);
|