pi-plans 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +40 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +13 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +157 -56
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +56 -10
- package/src/exec.ts +614 -126
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +13 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +39 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +18 -5
- package/tests/auditor.test.ts +116 -17
- package/tests/dashboard.test.ts +135 -1
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +65 -0
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
package/src/plan.ts
CHANGED
|
@@ -503,7 +503,7 @@ export function lintPlanTasks(planText: string): string | null {
|
|
|
503
503
|
}
|
|
504
504
|
}
|
|
505
505
|
// Covers references must point at known tasks; a check with zero covers
|
|
506
|
-
// never enters the
|
|
506
|
+
// never enters the execution review — surface that exemption loudly.
|
|
507
507
|
for (const item of parseChecklist(planText)) {
|
|
508
508
|
const covered = extractTaskCoverage(item.text);
|
|
509
509
|
if (covered.length === 0) {
|
package/src/refine-ui-state.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
|
|
2
2
|
|
|
3
|
-
export type RefineOverlayRole = "reviewer" | "refs";
|
|
3
|
+
export type RefineOverlayRole = "reviewer" | "refs" | "auditor";
|
|
4
4
|
export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
|
|
5
5
|
export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
|
|
6
6
|
|
package/src/refine-ui.ts
CHANGED
|
@@ -135,10 +135,12 @@ function previewTranscriptText(entry: RefineTranscriptEntry, width: number): { l
|
|
|
135
135
|
return { lines: lines.slice(-STREAMING_PREVIEW_LINES), truncated: true };
|
|
136
136
|
}
|
|
137
137
|
|
|
138
|
-
function footerText(laneCount: number, lang: UiLanguage): string {
|
|
138
|
+
function footerText(laneCount: number, lang: UiLanguage, role: RefineOverlayRole = "reviewer"): string {
|
|
139
139
|
const chrome = refineChrome(lang);
|
|
140
140
|
const parts = [chrome.close, chrome.scroll, chrome.page];
|
|
141
141
|
if (laneCount > 1) parts.push(chrome.switchLane);
|
|
142
|
+
// Auditor overlay only: the reopen shortcut is the way back after ESC.
|
|
143
|
+
if (role === "auditor") parts.push(chrome.reopen);
|
|
142
144
|
return parts.join(" · ");
|
|
143
145
|
}
|
|
144
146
|
|
|
@@ -166,7 +168,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
|
|
|
166
168
|
const complete = lanes.filter((lane) => lane.status === "complete").length;
|
|
167
169
|
const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
|
|
168
170
|
const running = lanes.filter((lane) => lane.status === "running").length;
|
|
169
|
-
const title = role === "reviewer" ? "Reviewer" : "Refs";
|
|
171
|
+
const title = role === "reviewer" ? "Reviewer" : role === "auditor" ? "Execution review" : "Refs";
|
|
170
172
|
const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
|
|
171
173
|
const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
|
|
172
174
|
return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
|
|
@@ -250,7 +252,7 @@ export class RefineOverlayComponent implements Component {
|
|
|
250
252
|
lines.push(...this.renderPane(this.lanes[index]!, innerWidth, paneHeight, index === this.selectedLane));
|
|
251
253
|
}
|
|
252
254
|
}
|
|
253
|
-
lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang)), innerWidth));
|
|
255
|
+
lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang, this.role)), innerWidth));
|
|
254
256
|
lines.push(renderBorderLine(this.theme, innerWidth, "bottom"));
|
|
255
257
|
return lines.map((line) => fitLine(line, width));
|
|
256
258
|
}
|
|
@@ -352,6 +354,20 @@ export class RefineOverlayController {
|
|
|
352
354
|
.catch(() => undefined);
|
|
353
355
|
}
|
|
354
356
|
|
|
357
|
+
/** Repaint without a progress event (the engine mutates lane state directly). */
|
|
358
|
+
rerender(): void {
|
|
359
|
+
if (this.closed) return;
|
|
360
|
+
this.tui?.requestRender();
|
|
361
|
+
}
|
|
362
|
+
|
|
363
|
+
/** Replace the lane with a matching id with engine-held state — used by the
|
|
364
|
+
* execution-review reopen path so a fresh controller continues the SAME
|
|
365
|
+
* transcript (one-shot controllers can never re-open themselves). */
|
|
366
|
+
seedLane(state: RefineLaneState): void {
|
|
367
|
+
const index = this.lanes.findIndex((lane) => lane.id === state.id);
|
|
368
|
+
if (index >= 0) this.lanes[index] = state;
|
|
369
|
+
}
|
|
370
|
+
|
|
355
371
|
update(laneId: string, event: SubagentProgressEvent): void {
|
|
356
372
|
if (this.closed) return;
|
|
357
373
|
const lane = this.lanes.find((candidate) => candidate.id === laneId);
|
package/src/resume-command.ts
CHANGED
|
@@ -300,11 +300,21 @@ async function buildBrief(
|
|
|
300
300
|
const reverify = load.reverifyAll
|
|
301
301
|
? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
|
|
302
302
|
: `\nPreviously verified and still valid: ${doneList}.`;
|
|
303
|
-
|
|
303
|
+
// v0.8: a review-cap pause is NOT cleared by this resume — only
|
|
304
|
+
// /plans-execute (an explicit user confirmation) grants a fresh
|
|
305
|
+
// five-round budget; ordinary resumes and input keep it paused.
|
|
306
|
+
const paused = load.pausedReason
|
|
307
|
+
? load.pausedReason.startsWith("execution review exhausted") || load.pausedReason.startsWith("completion audit exhausted")
|
|
308
|
+
? `\nExecution had been paused: ${load.pausedReason} — this pause survives the resume; run /plans-execute to grant a fresh five-round review budget.`
|
|
309
|
+
: `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.`
|
|
310
|
+
: "";
|
|
304
311
|
const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
|
|
312
|
+
// v0.8: a verifying run keeps checkpoint phase "executing" but the run
|
|
313
|
+
// STATUS is verifying — surface which loop owns the run right now.
|
|
314
|
+
const verifying = run.status === "verifying";
|
|
305
315
|
return {
|
|
306
|
-
phaseLabel: "executing",
|
|
307
|
-
text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\
|
|
316
|
+
phaseLabel: verifying ? "verifying" : "executing",
|
|
317
|
+
text: `[PI-PLANS RESUME] ${verifying ? "Execution review of" : "Execution of"} run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\n${verifying ? "The task tree is terminal and the execution-review loop owns the run: when all tasks are terminal and checks are still owed, a read-only reviewer round runs automatically (status verifying → done when every check passes). If a check fails, its tasks roll back to pending — fix and re-close them with plans_update_task." : "Follow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the execution reviewer verify the checks. The current wave and remaining tasks are injected each turn."}`,
|
|
308
318
|
};
|
|
309
319
|
}
|
|
310
320
|
if (load.legacyDelegate) {
|
package/src/resume.ts
CHANGED
|
@@ -27,7 +27,7 @@ export interface ResumeCandidate {
|
|
|
27
27
|
updatedAt: string;
|
|
28
28
|
}
|
|
29
29
|
|
|
30
|
-
const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "stopped"]);
|
|
30
|
+
const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "verifying", "stopped"]);
|
|
31
31
|
|
|
32
32
|
/** Terminal runs are resumable only when unfinished implementation-review evidence exists (D-008). */
|
|
33
33
|
function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null): boolean {
|
|
@@ -52,6 +52,10 @@ function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null
|
|
|
52
52
|
}
|
|
53
53
|
|
|
54
54
|
function phaseLabelOf(run: RunInfo, checkpoint: WorkflowCheckpoint | null): string {
|
|
55
|
+
// A verifying run keeps checkpoint.phase === "executing" (the execution
|
|
56
|
+
// state machine's phase is unchanged); surface the run status instead so
|
|
57
|
+
// the resume list never hides the verification loop behind "executing".
|
|
58
|
+
if (run.status === "verifying") return "verifying";
|
|
55
59
|
if (checkpoint !== null) return checkpoint.phase;
|
|
56
60
|
if (run.status === "stopped") return "executing (stopped)";
|
|
57
61
|
return run.status;
|
package/src/staleness.ts
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Stale-extension probe (v0.7.1, extracted).
|
|
3
|
+
*
|
|
4
|
+
* pi imports an extension's module graph once per process, so a fix written
|
|
5
|
+
* to disk while the session is running stays invisible until /reload. Both
|
|
6
|
+
* /plans (which reports it) and the execution reviewer (which suggests it when
|
|
7
|
+
* a verdict cannot be read) need the same answer, so the probe lives here
|
|
8
|
+
* rather than in index.ts -- exec.ts already depends on index.ts's imports and
|
|
9
|
+
* a back-import would close a cycle.
|
|
10
|
+
*
|
|
11
|
+
* index.ts records when its own copy was loaded; this module never holds that
|
|
12
|
+
* state, so callers pass it in and stay testable.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import * as fs from "node:fs";
|
|
16
|
+
import * as path from "node:path";
|
|
17
|
+
|
|
18
|
+
/** mtime (ms) of the newest .ts file under the extension root, or 0. */
|
|
19
|
+
export function newestSourceMtime(baseDir: string): number {
|
|
20
|
+
const stack = [baseDir, path.join(baseDir, "src"), path.join(baseDir, "tools")];
|
|
21
|
+
let newest = 0;
|
|
22
|
+
while (stack.length) {
|
|
23
|
+
const dir = stack.pop()!;
|
|
24
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
25
|
+
const full = path.join(dir, entry.name);
|
|
26
|
+
if (entry.isDirectory()) stack.push(full);
|
|
27
|
+
else if (entry.isFile() && entry.name.endsWith(".ts")) {
|
|
28
|
+
const mtime = fs.statSync(full).mtimeMs;
|
|
29
|
+
if (mtime > newest) newest = mtime;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
return newest;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** One line describing whether the loaded copy is current. A clock-skew
|
|
37
|
+
* tolerance keeps a same-second write from reading as staleness. */
|
|
38
|
+
export function stalenessLine(baseDir: string, loadedAt: Date): string {
|
|
39
|
+
try {
|
|
40
|
+
if (newestSourceMtime(baseDir) > loadedAt.getTime() + 2000) {
|
|
41
|
+
return `⚠ extension code on disk is newer than the loaded copy (loaded ${loadedAt.toISOString()}); run /reload to pick it up`;
|
|
42
|
+
}
|
|
43
|
+
return `Extension loaded: ${loadedAt.toISOString()} (up to date)`;
|
|
44
|
+
} catch {
|
|
45
|
+
return `Extension loaded: ${loadedAt.toISOString()}`;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Short form for inline hints: the /reload advice, or null when current. */
|
|
50
|
+
export function staleReloadHint(baseDir: string, loadedAt: Date): string | null {
|
|
51
|
+
const line = stalenessLine(baseDir, loadedAt);
|
|
52
|
+
return line.startsWith("⚠") ? line : null;
|
|
53
|
+
}
|
package/src/state.ts
CHANGED
package/src/task-tool.ts
CHANGED
|
@@ -64,7 +64,7 @@ export function registerTaskStatusTool(ext: ExtensionAPI): void {
|
|
|
64
64
|
name: "plans_update_task",
|
|
65
65
|
label: "Update task",
|
|
66
66
|
description:
|
|
67
|
-
'Report execution progress for one task of the accepted plan: set status "complete" (with evidence) or "skipped" (with skipReason). Fails outside pi-plans execution mode. Statuses are immutable once set — the independent
|
|
67
|
+
'Report execution progress for one task of the accepted plan: set status "complete" (with evidence) or "skipped" (with skipReason). Fails outside pi-plans execution mode. Statuses are immutable once set — the independent execution reviewer handles any rollback.',
|
|
68
68
|
promptSnippet: "Report plan task completion",
|
|
69
69
|
parameters: UpdateTaskParams,
|
|
70
70
|
|
package/src/tasks.ts
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
|
|
3
3
|
* (`## Tasks`) as the unit of progress. Status flows in exclusively through
|
|
4
4
|
* the task status tool; completion of the whole run is gated by the
|
|
5
|
-
* independent
|
|
5
|
+
* independent execution reviewer over the plan's verification checks.
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
|
|
@@ -77,11 +77,13 @@ export function allTasksTerminal(tasks: TaskView[]): boolean {
|
|
|
77
77
|
return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
|
|
78
78
|
}
|
|
79
79
|
|
|
80
|
-
/** Snapshot of statuses for persistence and stall detection.
|
|
80
|
+
/** Snapshot of statuses for persistence and stall detection. Every task gets
|
|
81
|
+
* a record: a rolled-back task must stay visible with the evidence from its
|
|
82
|
+
* previous attempt, otherwise a rollback erases the run's history and the
|
|
83
|
+
* agent can no longer tell "done and rolled back" from "never started". */
|
|
81
84
|
export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
|
|
82
85
|
const map: TaskProgressMap = {};
|
|
83
86
|
for (const task of flattenTaskViews(tasks)) {
|
|
84
|
-
if (task.status === "pending" && task.evidence === undefined && task.skipReason === undefined) continue;
|
|
85
87
|
map[task.id] = {
|
|
86
88
|
status: task.status,
|
|
87
89
|
evidence: task.evidence,
|
|
@@ -132,7 +134,13 @@ export function canTransition(task: TaskView, next: TaskStatus): boolean {
|
|
|
132
134
|
|
|
133
135
|
/** Rollback set for a failed verification check: every task in its covers
|
|
134
136
|
* clause (parents cascade to their children, skipped tasks reopen too).
|
|
135
|
-
* Returns the ids that actually reopen.
|
|
137
|
+
* Returns the ids that actually reopen.
|
|
138
|
+
*
|
|
139
|
+
* The task's `evidence` is deliberately kept. It records what the previous
|
|
140
|
+
* attempt actually did, which is the one thing the agent cannot reconstruct
|
|
141
|
+
* once it reopens the task; re-reporting overwrites it. `skipReason` is
|
|
142
|
+
* cleared because it described a deliberate skip that the audit has now
|
|
143
|
+
* overturned. */
|
|
136
144
|
export function auditRollbackSet(
|
|
137
145
|
tasks: TaskView[],
|
|
138
146
|
checklist: CheckItem[],
|
|
@@ -146,7 +154,6 @@ export function auditRollbackSet(
|
|
|
146
154
|
const reopenNode = (node: TaskView): void => {
|
|
147
155
|
if (node.status !== "pending") {
|
|
148
156
|
node.status = "pending";
|
|
149
|
-
node.evidence = undefined;
|
|
150
157
|
node.skipReason = undefined;
|
|
151
158
|
reopen.push(node.id);
|
|
152
159
|
}
|
|
@@ -158,6 +165,33 @@ export function auditRollbackSet(
|
|
|
158
165
|
return reopen;
|
|
159
166
|
}
|
|
160
167
|
|
|
168
|
+
/** Checks that lose their satisfied state because a rollback reopened work
|
|
169
|
+
* they were verifying. Returns the ids whose `done` flag was cleared.
|
|
170
|
+
*
|
|
171
|
+
* A check can be presolved without an audit round (skipped-pass), or affirmed
|
|
172
|
+
* in an earlier round, and stay `done` while another check's rollback reopens
|
|
173
|
+
* one of its covered tasks. Left alone it renders as "check passed" next to a
|
|
174
|
+
* task that is open again with stale evidence. The caller owns this: it runs
|
|
175
|
+
* after the rollback set is known. */
|
|
176
|
+
export function invalidateChecksForRolledBackTasks(
|
|
177
|
+
checklist: CheckItem[],
|
|
178
|
+
tasks: TaskView[],
|
|
179
|
+
reopenedIds: string[],
|
|
180
|
+
): string[] {
|
|
181
|
+
if (reopenedIds.length === 0) return [];
|
|
182
|
+
const reopened = new Set(reopenedIds);
|
|
183
|
+
const cleared: string[] = [];
|
|
184
|
+
for (const item of checklist) {
|
|
185
|
+
if (!item.done) continue;
|
|
186
|
+
const covered = extractTaskCoverage(item.text);
|
|
187
|
+
if (covered.some((id) => reopened.has(id))) {
|
|
188
|
+
item.done = false;
|
|
189
|
+
cleared.push(item.id);
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
return cleared;
|
|
193
|
+
}
|
|
194
|
+
|
|
161
195
|
/** Verification checks that cover no task are excluded from audit (their
|
|
162
196
|
* pass state cannot be derived from task statuses). */
|
|
163
197
|
export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
|
package/src/ui-language.ts
CHANGED
|
@@ -103,6 +103,8 @@ export interface RefineChrome {
|
|
|
103
103
|
scroll: string;
|
|
104
104
|
page: string;
|
|
105
105
|
switchLane: string;
|
|
106
|
+
/** Auditor overlay only (v0.8): the shortcut that reopens the in-flight round. */
|
|
107
|
+
reopen: string;
|
|
106
108
|
}
|
|
107
109
|
|
|
108
110
|
const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
|
|
@@ -111,12 +113,14 @@ const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
|
|
|
111
113
|
scroll: "↑/↓ 滚动",
|
|
112
114
|
page: "PgUp/PgDn 翻页",
|
|
113
115
|
switchLane: "Tab & Shift + Tab 切换 lane",
|
|
116
|
+
reopen: "Ctrl+Shift+R 重开评审面板",
|
|
114
117
|
},
|
|
115
118
|
en: {
|
|
116
119
|
close: "Esc close",
|
|
117
120
|
scroll: "↑/↓ scroll",
|
|
118
121
|
page: "PgUp/PgDn page",
|
|
119
122
|
switchLane: "Tab & Shift + Tab switch lane",
|
|
123
|
+
reopen: "Ctrl+Shift+R reopen review overlay",
|
|
120
124
|
},
|
|
121
125
|
};
|
|
122
126
|
|
package/src/workflow-state.ts
CHANGED
|
@@ -152,8 +152,12 @@ export interface ExecutionCheckpoint {
|
|
|
152
152
|
* progress record. doneVcIds stays for legacy checkpoints and the final
|
|
153
153
|
* audit pass. */
|
|
154
154
|
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
155
|
+
/** v0.7.1: watchdog round counter. Persisted so a session restart cannot
|
|
156
|
+
* silently hand a stalled run a fresh budget; optional so checkpoints
|
|
157
|
+
* written before this field keep loading. */
|
|
158
|
+
stallRounds?: number;
|
|
155
159
|
/** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
|
|
156
|
-
audit?: { rounds: number; lastResult?: string; passed?: boolean };
|
|
160
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
|
|
157
161
|
}
|
|
158
162
|
|
|
159
163
|
export interface OwnerInfo {
|
|
@@ -461,7 +465,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
461
465
|
const record = asRecord(value, label);
|
|
462
466
|
rejectExtraKeys(
|
|
463
467
|
record,
|
|
464
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "audit"]),
|
|
468
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
|
|
465
469
|
label,
|
|
466
470
|
);
|
|
467
471
|
const execution: ExecutionCheckpoint = {
|
|
@@ -505,14 +509,20 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
505
509
|
}
|
|
506
510
|
execution.tasks = tasks;
|
|
507
511
|
}
|
|
512
|
+
if (record.stallRounds !== undefined && record.stallRounds !== null) {
|
|
513
|
+
execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
|
|
514
|
+
}
|
|
508
515
|
if (record.audit !== undefined && record.audit !== null) {
|
|
509
516
|
const audit = asRecord(record.audit, `${label}.audit`);
|
|
510
|
-
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed"]), `${label}.audit`);
|
|
511
|
-
const parsed: { rounds: number; lastResult?: string; passed?: boolean } = {
|
|
517
|
+
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
|
|
518
|
+
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
|
|
512
519
|
rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
|
|
513
520
|
};
|
|
514
521
|
if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
|
|
515
522
|
if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
|
|
523
|
+
if (audit.undeterminable !== undefined) {
|
|
524
|
+
parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
|
|
525
|
+
}
|
|
516
526
|
execution.audit = parsed;
|
|
517
527
|
}
|
|
518
528
|
return execution;
|
|
@@ -1071,7 +1081,9 @@ export interface ExecutionProgressInput {
|
|
|
1071
1081
|
/** v0.6.1: task-tree progress snapshot (authoritative). */
|
|
1072
1082
|
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
1073
1083
|
/** v0.6.1: completion-audit bookkeeping update. */
|
|
1074
|
-
audit?: { rounds: number; lastResult?: string; passed?: boolean };
|
|
1084
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
|
|
1085
|
+
/** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
|
|
1086
|
+
stallRounds?: number;
|
|
1075
1087
|
}
|
|
1076
1088
|
|
|
1077
1089
|
export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
|
|
@@ -1091,6 +1103,7 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1091
1103
|
if (progress.delegate === null) delete execution.delegate;
|
|
1092
1104
|
else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
|
|
1093
1105
|
if (progress.tasks !== undefined) execution.tasks = progress.tasks;
|
|
1106
|
+
if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
|
|
1094
1107
|
if (progress.audit !== undefined) execution.audit = progress.audit;
|
|
1095
1108
|
return { ...cp, execution };
|
|
1096
1109
|
}
|
package/tests/auditor.test.ts
CHANGED
|
@@ -1,11 +1,17 @@
|
|
|
1
|
-
/** Tests for the
|
|
2
|
-
* boundaries, skipped-pass, and no-cover exclusion. */
|
|
1
|
+
/** Tests for the execution reviewer: tri-state verdict parsing, contract
|
|
2
|
+
* coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
|
|
3
3
|
|
|
4
4
|
import * as assert from "node:assert/strict";
|
|
5
5
|
import { describe, it } from "node:test";
|
|
6
6
|
import type { CheckItem } from "../src/plan.ts";
|
|
7
|
-
import { buildTaskView } from "../src/tasks.ts";
|
|
8
|
-
import {
|
|
7
|
+
import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
|
|
8
|
+
import {
|
|
9
|
+
applyAuditOutcome,
|
|
10
|
+
auditablePendingChecks,
|
|
11
|
+
buildAuditTask,
|
|
12
|
+
parseAuditReport,
|
|
13
|
+
presolvedCheckIds,
|
|
14
|
+
} from "../src/auditor.ts";
|
|
9
15
|
import { parsePlanTasks } from "../src/plan.ts";
|
|
10
16
|
|
|
11
17
|
const PLAN = `## Tasks
|
|
@@ -37,6 +43,8 @@ function view(progress?: Record<string, { status: "complete" | "skipped" }>) {
|
|
|
37
43
|
return buildTaskView(parsePlanTasks(PLAN), progress);
|
|
38
44
|
}
|
|
39
45
|
|
|
46
|
+
const ALL_IDS = ["VC-001", "VC-002", "VC-003", "VC-004"];
|
|
47
|
+
|
|
40
48
|
describe("auditor parsing", () => {
|
|
41
49
|
it("parses per-check verdicts and ignores unknown ids", () => {
|
|
42
50
|
const report = [
|
|
@@ -44,48 +52,139 @@ describe("auditor parsing", () => {
|
|
|
44
52
|
"- `VC-002` — verdict: fail; evidence: src/c.ts; note: missing",
|
|
45
53
|
"- `VC-999` — verdict: pass; evidence: n/a",
|
|
46
54
|
].join("\n");
|
|
47
|
-
const { passed, failed } = parseAuditReport(report,
|
|
55
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
48
56
|
assert.deepEqual(passed, ["VC-001"]);
|
|
49
57
|
assert.deepEqual(failed, ["VC-002"]);
|
|
58
|
+
assert.deepEqual(undeterminable.sort(), ["VC-003", "VC-004"]);
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
it("parses verdicts wrapped in markdown emphasis", () => {
|
|
62
|
+
// Regression: `verdict: **pass**` used to parse as nothing, and the
|
|
63
|
+
// caller's fail-closed rule then rolled the entire run back.
|
|
64
|
+
const report = [
|
|
65
|
+
"- `VC-001` — verdict: **pass**; evidence: src/a.ts",
|
|
66
|
+
"- `VC-002` — verdict: **fail**; evidence: src/c.ts",
|
|
67
|
+
"- `VC-003` — verdict: *pass*; evidence: src/d.ts",
|
|
68
|
+
"- `VC-004` — verdict: `pass`; evidence: src/e.ts",
|
|
69
|
+
].join("\n");
|
|
70
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
71
|
+
assert.deepEqual(passed, ["VC-001", "VC-003", "VC-004"]);
|
|
72
|
+
assert.deepEqual(failed, ["VC-002"]);
|
|
73
|
+
assert.deepEqual(undeterminable, []);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
it("does not read a verdict word prefix as a verdict", () => {
|
|
77
|
+
const report = ["- `VC-001` — verdict: passed; evidence: x", "- `VC-002` — verdict: failing; evidence: y"].join("\n");
|
|
78
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
79
|
+
assert.deepEqual(passed, []);
|
|
80
|
+
assert.deepEqual(failed, []);
|
|
81
|
+
assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
|
|
50
82
|
});
|
|
51
83
|
|
|
52
84
|
it("conflicting verdicts for one check resolve to fail", () => {
|
|
53
85
|
const report = "- `VC-001` — verdict: pass; note: a\n- `VC-001` — verdict: fail; note: b";
|
|
54
|
-
const { passed, failed } = parseAuditReport(report,
|
|
86
|
+
const { passed, failed } = parseAuditReport(report, ALL_IDS);
|
|
55
87
|
assert.deepEqual(passed, []);
|
|
56
88
|
assert.deepEqual(failed, ["VC-001"]);
|
|
57
89
|
});
|
|
58
90
|
|
|
59
|
-
it("
|
|
91
|
+
it("reads an explicit undeterminable verdict", () => {
|
|
92
|
+
const report = "- `VC-001` — verdict: pass\n- `VC-002` — verdict: undeterminable; evidence: bash not granted";
|
|
93
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
94
|
+
assert.deepEqual(passed, ["VC-001"]);
|
|
95
|
+
assert.deepEqual(failed, []);
|
|
96
|
+
assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-003", "VC-004"]);
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
it("treats a reviewer-shaped report as undeterminable, never as failure", () => {
|
|
100
|
+
// Regression: agents/reviewer.md was the auditor's system prompt and
|
|
101
|
+
// mandates `## Findings` / `## Questions` with F-### lines and never
|
|
102
|
+
// says "verdict". Under fail-closed that whole shape read as "every
|
|
103
|
+
// check failed" and rolled the run back.
|
|
104
|
+
const report = [
|
|
105
|
+
"## Findings",
|
|
106
|
+
"",
|
|
107
|
+
"- `F-001` — severity: high; affected: VC-001; evidence: src/a.ts; impact: x; recommended fix: y; suggested disposition: accept.",
|
|
108
|
+
"",
|
|
109
|
+
"## Questions",
|
|
110
|
+
"",
|
|
111
|
+
"None.",
|
|
112
|
+
].join("\n");
|
|
113
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
114
|
+
assert.deepEqual(passed, []);
|
|
115
|
+
assert.deepEqual(failed, [], "a reviewer-shaped report is not a failed audit");
|
|
116
|
+
assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
it("classifies a partially covered report per check", () => {
|
|
120
|
+
const report = "- `VC-001` — verdict: pass\n- `VC-003` — verdict: **pass**";
|
|
121
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
122
|
+
assert.deepEqual(passed, ["VC-001", "VC-003"]);
|
|
123
|
+
assert.deepEqual(failed, []);
|
|
124
|
+
assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-004"]);
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
it("builds the brief with the auditable pending checks only", () => {
|
|
60
128
|
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
61
129
|
assert.match(task, /read-only/i);
|
|
62
130
|
assert.match(task, /`VC-001`/);
|
|
63
131
|
assert.doesNotMatch(task, /`VC-999`/);
|
|
132
|
+
assert.doesNotMatch(task, /`VC-004`/, "a check covering no task is never audited");
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
it("brief states the tri-state contract and forbids fail-for-want-of-evidence", () => {
|
|
136
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
137
|
+
assert.match(task, /verdict: pass \| fail \| undeterminable/);
|
|
138
|
+
assert.match(task, /never report `fail` for want of evidence/i);
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
it("brief omits checks already done in an earlier round", () => {
|
|
142
|
+
// Regression: the brief listed every auditable check, so a round-2
|
|
143
|
+
// auditor that legitimately skipped the satisfied ones looked like a
|
|
144
|
+
// contract violation -- and, with no rollback to produce progress, the
|
|
145
|
+
// budget climbed to the cap and the run livelocked.
|
|
146
|
+
const done = buildAuditTask("/tmp/PLAN_v1.md", checks(["VC-001"]), view(), 2);
|
|
147
|
+
assert.doesNotMatch(done, /`VC-001`/);
|
|
148
|
+
assert.match(done, /`VC-002`/);
|
|
149
|
+
assert.deepEqual(
|
|
150
|
+
auditablePendingChecks(checks(["VC-001"]), view()).map((item) => item.id),
|
|
151
|
+
["VC-002", "VC-003"],
|
|
152
|
+
);
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
it("applyAuditOutcome classifies without touching the checklist or task tree", () => {
|
|
156
|
+
const checklist = checks();
|
|
157
|
+
const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
158
|
+
const parsed = parseAuditReport("- `VC-001` — verdict: pass\n- `VC-002` — verdict: fail", ["VC-001", "VC-002"]);
|
|
159
|
+
const outcome = applyAuditOutcome(3, parsed, "report text");
|
|
160
|
+
assert.deepEqual(outcome.passed, ["VC-001"]);
|
|
161
|
+
assert.deepEqual(outcome.failed, ["VC-002"]);
|
|
162
|
+
assert.deepEqual(outcome.undeterminable, []);
|
|
163
|
+
assert.equal(outcome.round, 3);
|
|
164
|
+
assert.equal(outcome.report, "report text");
|
|
165
|
+
assert.equal(checklist.every((item) => item.done === false), true, "no done flag written here");
|
|
166
|
+
assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "complete", "no rollback here");
|
|
64
167
|
});
|
|
65
168
|
});
|
|
66
169
|
|
|
67
170
|
describe("audit rollback boundaries", () => {
|
|
68
171
|
it("failed multi-cover check rolls back every covered task", () => {
|
|
69
172
|
const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
70
|
-
const
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
// Passed check stays done; its task stays closed.
|
|
74
|
-
assert.equal(checks().length, 4);
|
|
173
|
+
const rolledBack = auditRollbackSet(tasks, checks(), "VC-002");
|
|
174
|
+
assert.deepEqual(rolledBack.sort(), ["Task-2", "Task-3", "Task-3.1"]);
|
|
175
|
+
// A check that passed keeps its task closed.
|
|
75
176
|
assert.equal(tasks.find((t) => t.id === "Task-1")?.status, "complete");
|
|
76
177
|
assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "pending");
|
|
77
178
|
});
|
|
78
179
|
|
|
79
180
|
it("covering the parent cascades to subtasks even when the parent alone is listed", () => {
|
|
80
181
|
const tasks = view({ "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
81
|
-
|
|
82
|
-
assert.ok(outcome.rolledBack.includes("Task-3.1"), "subtask reopens with the parent");
|
|
182
|
+
assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-3.1"), "subtask reopens with the parent");
|
|
83
183
|
});
|
|
84
184
|
|
|
85
185
|
it("skipped tasks reopen too (skip state cleared)", () => {
|
|
86
186
|
const tasks = view({ "Task-2": { status: "skipped" } });
|
|
87
|
-
|
|
88
|
-
assert.ok(outcome.rolledBack.includes("Task-2"));
|
|
187
|
+
assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-2"));
|
|
89
188
|
assert.equal(tasks.find((t) => t.id === "Task-2")?.skipReason, undefined);
|
|
90
189
|
});
|
|
91
190
|
|
|
@@ -108,4 +207,4 @@ describe("audit rollback boundaries", () => {
|
|
|
108
207
|
assert.ok(presolved.includes("VC-003"));
|
|
109
208
|
assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
|
|
110
209
|
});
|
|
111
|
-
});
|
|
210
|
+
});
|