pi-plans 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/plan.ts CHANGED
@@ -503,7 +503,7 @@ export function lintPlanTasks(planText: string): string | null {
503
503
  }
504
504
  }
505
505
  // Covers references must point at known tasks; a check with zero covers
506
- // never enters the completion audit — surface that exemption loudly.
506
+ // never enters the execution review — surface that exemption loudly.
507
507
  for (const item of parseChecklist(planText)) {
508
508
  const covered = extractTaskCoverage(item.text);
509
509
  if (covered.length === 0) {
@@ -1,6 +1,6 @@
1
1
  import type { SubagentProgressEvent, SubagentResult } from "./subagent.ts";
2
2
 
3
- export type RefineOverlayRole = "reviewer" | "refs";
3
+ export type RefineOverlayRole = "reviewer" | "refs" | "auditor";
4
4
  export type RefineLaneStatus = "queued" | "running" | "complete" | "failed" | "cancelled";
5
5
  export type RefineTranscriptEntryType = "assistant-text" | "thinking" | "tool-call" | "tool-result" | "diagnostic";
6
6
 
package/src/refine-ui.ts CHANGED
@@ -135,10 +135,12 @@ function previewTranscriptText(entry: RefineTranscriptEntry, width: number): { l
135
135
  return { lines: lines.slice(-STREAMING_PREVIEW_LINES), truncated: true };
136
136
  }
137
137
 
138
- function footerText(laneCount: number, lang: UiLanguage): string {
138
+ function footerText(laneCount: number, lang: UiLanguage, role: RefineOverlayRole = "reviewer"): string {
139
139
  const chrome = refineChrome(lang);
140
140
  const parts = [chrome.close, chrome.scroll, chrome.page];
141
141
  if (laneCount > 1) parts.push(chrome.switchLane);
142
+ // Auditor overlay only: the reopen shortcut is the way back after ESC.
143
+ if (role === "auditor") parts.push(chrome.reopen);
142
144
  return parts.join(" · ");
143
145
  }
144
146
 
@@ -166,7 +168,7 @@ function summaryFor(role: RefineOverlayRole, lanes: RefineLaneState[], modelLabe
166
168
  const complete = lanes.filter((lane) => lane.status === "complete").length;
167
169
  const terminal = lanes.filter((lane) => ["complete", "failed", "cancelled"].includes(lane.status)).length;
168
170
  const running = lanes.filter((lane) => lane.status === "running").length;
169
- const title = role === "reviewer" ? "Reviewer" : "Refs";
171
+ const title = role === "reviewer" ? "Reviewer" : role === "auditor" ? "Execution review" : "Refs";
170
172
  const visibleTitle = modelLabel ? `${title} (${modelLabel})` : title;
171
173
  const state = terminal === lanes.length ? "done" : running > 0 ? `${running} running` : "queued";
172
174
  return `${visibleTitle} · ${complete}/${lanes.length} done · ${state}`;
@@ -250,7 +252,7 @@ export class RefineOverlayComponent implements Component {
250
252
  lines.push(...this.renderPane(this.lanes[index]!, innerWidth, paneHeight, index === this.selectedLane));
251
253
  }
252
254
  }
253
- lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang)), innerWidth));
255
+ lines.push(renderRow(this.theme, this.theme.fg("dim", footerText(this.lanes.length, this.lang, this.role)), innerWidth));
254
256
  lines.push(renderBorderLine(this.theme, innerWidth, "bottom"));
255
257
  return lines.map((line) => fitLine(line, width));
256
258
  }
@@ -352,6 +354,20 @@ export class RefineOverlayController {
352
354
  .catch(() => undefined);
353
355
  }
354
356
 
357
+ /** Repaint without a progress event (the engine mutates lane state directly). */
358
+ rerender(): void {
359
+ if (this.closed) return;
360
+ this.tui?.requestRender();
361
+ }
362
+
363
+ /** Replace the lane with a matching id with engine-held state — used by the
364
+ * execution-review reopen path so a fresh controller continues the SAME
365
+ * transcript (one-shot controllers can never re-open themselves). */
366
+ seedLane(state: RefineLaneState): void {
367
+ const index = this.lanes.findIndex((lane) => lane.id === state.id);
368
+ if (index >= 0) this.lanes[index] = state;
369
+ }
370
+
355
371
  update(laneId: string, event: SubagentProgressEvent): void {
356
372
  if (this.closed) return;
357
373
  const lane = this.lanes.find((candidate) => candidate.id === laneId);
@@ -300,11 +300,21 @@ async function buildBrief(
300
300
  const reverify = load.reverifyAll
301
301
  ? `\nThe code state (HEAD) changed since approval: the authorization is KEPT, but every previously closed task was re-opened and must be re-done. Historically verified checks (evidence only): ${doneList}.`
302
302
  : `\nPreviously verified and still valid: ${doneList}.`;
303
- const paused = load.pausedReason ? `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.` : "";
303
+ // v0.8: a review-cap pause is NOT cleared by this resume — only
304
+ // /plans-execute (an explicit user confirmation) grants a fresh
305
+ // five-round budget; ordinary resumes and input keep it paused.
306
+ const paused = load.pausedReason
307
+ ? load.pausedReason.startsWith("execution review exhausted") || load.pausedReason.startsWith("completion audit exhausted")
308
+ ? `\nExecution had been paused: ${load.pausedReason} — this pause survives the resume; run /plans-execute to grant a fresh five-round review budget.`
309
+ : `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.`
310
+ : "";
304
311
  const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
312
+ // v0.8: a verifying run keeps checkpoint phase "executing" but the run
313
+ // STATUS is verifying — surface which loop owns the run right now.
314
+ const verifying = run.status === "verifying";
305
315
  return {
306
- phaseLabel: "executing",
307
- text: `[PI-PLANS RESUME] Execution of run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\nFollow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the completion auditor verify the checks. The current wave and remaining tasks are injected each turn.`,
316
+ phaseLabel: verifying ? "verifying" : "executing",
317
+ text: `[PI-PLANS RESUME] ${verifying ? "Execution review of" : "Execution of"} run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\n${verifying ? "The task tree is terminal and the execution-review loop owns the run: when all tasks are terminal and checks are still owed, a read-only reviewer round runs automatically (status verifying → done when every check passes). If a check fails, its tasks roll back to pending — fix and re-close them with plans_update_task." : "Follow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the execution reviewer verify the checks. The current wave and remaining tasks are injected each turn."}`,
308
318
  };
309
319
  }
310
320
  if (load.legacyDelegate) {
package/src/resume.ts CHANGED
@@ -27,7 +27,7 @@ export interface ResumeCandidate {
27
27
  updatedAt: string;
28
28
  }
29
29
 
30
- const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "stopped"]);
30
+ const RESUMABLE_RUN_STATUSES = new Set(["planning", "accepted", "executing", "verifying", "stopped"]);
31
31
 
32
32
  /** Terminal runs are resumable only when unfinished implementation-review evidence exists (D-008). */
33
33
  function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null): boolean {
@@ -52,6 +52,10 @@ function legacyDoneResumable(run: RunInfo, checkpoint: WorkflowCheckpoint | null
52
52
  }
53
53
 
54
54
  function phaseLabelOf(run: RunInfo, checkpoint: WorkflowCheckpoint | null): string {
55
+ // A verifying run keeps checkpoint.phase === "executing" (the execution
56
+ // state machine's phase is unchanged); surface the run status instead so
57
+ // the resume list never hides the verification loop behind "executing".
58
+ if (run.status === "verifying") return "verifying";
55
59
  if (checkpoint !== null) return checkpoint.phase;
56
60
  if (run.status === "stopped") return "executing (stopped)";
57
61
  return run.status;
@@ -0,0 +1,53 @@
1
+ /**
2
+ * Stale-extension probe (v0.7.1, extracted).
3
+ *
4
+ * pi imports an extension's module graph once per process, so a fix written
5
+ * to disk while the session is running stays invisible until /reload. Both
6
+ * /plans (which reports it) and the execution reviewer (which suggests it when
7
+ * a verdict cannot be read) need the same answer, so the probe lives here
8
+ * rather than in index.ts -- exec.ts already depends on index.ts's imports and
9
+ * a back-import would close a cycle.
10
+ *
11
+ * index.ts records when its own copy was loaded; this module never holds that
12
+ * state, so callers pass it in and stay testable.
13
+ */
14
+
15
+ import * as fs from "node:fs";
16
+ import * as path from "node:path";
17
+
18
+ /** mtime (ms) of the newest .ts file under the extension root, or 0. */
19
+ export function newestSourceMtime(baseDir: string): number {
20
+ const stack = [baseDir, path.join(baseDir, "src"), path.join(baseDir, "tools")];
21
+ let newest = 0;
22
+ while (stack.length) {
23
+ const dir = stack.pop()!;
24
+ for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
25
+ const full = path.join(dir, entry.name);
26
+ if (entry.isDirectory()) stack.push(full);
27
+ else if (entry.isFile() && entry.name.endsWith(".ts")) {
28
+ const mtime = fs.statSync(full).mtimeMs;
29
+ if (mtime > newest) newest = mtime;
30
+ }
31
+ }
32
+ }
33
+ return newest;
34
+ }
35
+
36
+ /** One line describing whether the loaded copy is current. A clock-skew
37
+ * tolerance keeps a same-second write from reading as staleness. */
38
+ export function stalenessLine(baseDir: string, loadedAt: Date): string {
39
+ try {
40
+ if (newestSourceMtime(baseDir) > loadedAt.getTime() + 2000) {
41
+ return `⚠ extension code on disk is newer than the loaded copy (loaded ${loadedAt.toISOString()}); run /reload to pick it up`;
42
+ }
43
+ return `Extension loaded: ${loadedAt.toISOString()} (up to date)`;
44
+ } catch {
45
+ return `Extension loaded: ${loadedAt.toISOString()}`;
46
+ }
47
+ }
48
+
49
+ /** Short form for inline hints: the /reload advice, or null when current. */
50
+ export function staleReloadHint(baseDir: string, loadedAt: Date): string | null {
51
+ const line = stalenessLine(baseDir, loadedAt);
52
+ return line.startsWith("⚠") ? line : null;
53
+ }
package/src/state.ts CHANGED
@@ -119,6 +119,7 @@ export const VALID_RUN_STATUSES = new Set([
119
119
  "planning",
120
120
  "accepted",
121
121
  "executing",
122
+ "verifying",
122
123
  "stopped",
123
124
  "abandoned",
124
125
  "done",
package/src/task-tool.ts CHANGED
@@ -64,7 +64,7 @@ export function registerTaskStatusTool(ext: ExtensionAPI): void {
64
64
  name: "plans_update_task",
65
65
  label: "Update task",
66
66
  description:
67
- 'Report execution progress for one task of the accepted plan: set status "complete" (with evidence) or "skipped" (with skipReason). Fails outside pi-plans execution mode. Statuses are immutable once set — the independent completion auditor handles any rollback.',
67
+ 'Report execution progress for one task of the accepted plan: set status "complete" (with evidence) or "skipped" (with skipReason). Fails outside pi-plans execution mode. Statuses are immutable once set — the independent execution reviewer handles any rollback.',
68
68
  promptSnippet: "Report plan task completion",
69
69
  parameters: UpdateTaskParams,
70
70
 
package/src/tasks.ts CHANGED
@@ -2,7 +2,7 @@
2
2
  * Task-tree runtime model (v0.6.1): the execution phase tracks plan tasks
3
3
  * (`## Tasks`) as the unit of progress. Status flows in exclusively through
4
4
  * the task status tool; completion of the whole run is gated by the
5
- * independent completion auditor over the plan's verification checks.
5
+ * independent execution reviewer over the plan's verification checks.
6
6
  */
7
7
 
8
8
  import type { CheckItem, PlanTasks, TaskNode, WaveEntry } from "./plan.ts";
@@ -77,11 +77,13 @@ export function allTasksTerminal(tasks: TaskView[]): boolean {
77
77
  return flattenTaskViews(tasks).length > 0 && flattenTaskViews(tasks).every((task) => taskIsTerminal(task));
78
78
  }
79
79
 
80
- /** Snapshot of statuses for persistence and stall detection. */
80
+ /** Snapshot of statuses for persistence and stall detection. Every task gets
81
+ * a record: a rolled-back task must stay visible with the evidence from its
82
+ * previous attempt, otherwise a rollback erases the run's history and the
83
+ * agent can no longer tell "done and rolled back" from "never started". */
81
84
  export function taskProgressMap(tasks: TaskView[]): TaskProgressMap {
82
85
  const map: TaskProgressMap = {};
83
86
  for (const task of flattenTaskViews(tasks)) {
84
- if (task.status === "pending" && task.evidence === undefined && task.skipReason === undefined) continue;
85
87
  map[task.id] = {
86
88
  status: task.status,
87
89
  evidence: task.evidence,
@@ -132,7 +134,13 @@ export function canTransition(task: TaskView, next: TaskStatus): boolean {
132
134
 
133
135
  /** Rollback set for a failed verification check: every task in its covers
134
136
  * clause (parents cascade to their children, skipped tasks reopen too).
135
- * Returns the ids that actually reopen. */
137
+ * Returns the ids that actually reopen.
138
+ *
139
+ * The task's `evidence` is deliberately kept. It records what the previous
140
+ * attempt actually did, which is the one thing the agent cannot reconstruct
141
+ * once it reopens the task; re-reporting overwrites it. `skipReason` is
142
+ * cleared because it described a deliberate skip that the audit has now
143
+ * overturned. */
136
144
  export function auditRollbackSet(
137
145
  tasks: TaskView[],
138
146
  checklist: CheckItem[],
@@ -146,7 +154,6 @@ export function auditRollbackSet(
146
154
  const reopenNode = (node: TaskView): void => {
147
155
  if (node.status !== "pending") {
148
156
  node.status = "pending";
149
- node.evidence = undefined;
150
157
  node.skipReason = undefined;
151
158
  reopen.push(node.id);
152
159
  }
@@ -158,6 +165,33 @@ export function auditRollbackSet(
158
165
  return reopen;
159
166
  }
160
167
 
168
+ /** Checks that lose their satisfied state because a rollback reopened work
169
+ * they were verifying. Returns the ids whose `done` flag was cleared.
170
+ *
171
+ * A check can be presolved without an audit round (skipped-pass), or affirmed
172
+ * in an earlier round, and stay `done` while another check's rollback reopens
173
+ * one of its covered tasks. Left alone it renders as "check passed" next to a
174
+ * task that is open again with stale evidence. The caller owns this: it runs
175
+ * after the rollback set is known. */
176
+ export function invalidateChecksForRolledBackTasks(
177
+ checklist: CheckItem[],
178
+ tasks: TaskView[],
179
+ reopenedIds: string[],
180
+ ): string[] {
181
+ if (reopenedIds.length === 0) return [];
182
+ const reopened = new Set(reopenedIds);
183
+ const cleared: string[] = [];
184
+ for (const item of checklist) {
185
+ if (!item.done) continue;
186
+ const covered = extractTaskCoverage(item.text);
187
+ if (covered.some((id) => reopened.has(id))) {
188
+ item.done = false;
189
+ cleared.push(item.id);
190
+ }
191
+ }
192
+ return cleared;
193
+ }
194
+
161
195
  /** Verification checks that cover no task are excluded from audit (their
162
196
  * pass state cannot be derived from task statuses). */
163
197
  export function auditableChecks(checklist: CheckItem[], tasks: TaskView[]): CheckItem[] {
@@ -103,6 +103,8 @@ export interface RefineChrome {
103
103
  scroll: string;
104
104
  page: string;
105
105
  switchLane: string;
106
+ /** Auditor overlay only (v0.8): the shortcut that reopens the in-flight round. */
107
+ reopen: string;
106
108
  }
107
109
 
108
110
  const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
@@ -111,12 +113,14 @@ const REFINE_CHROME: Record<UiLanguage, RefineChrome> = {
111
113
  scroll: "↑/↓ 滚动",
112
114
  page: "PgUp/PgDn 翻页",
113
115
  switchLane: "Tab & Shift + Tab 切换 lane",
116
+ reopen: "Ctrl+Shift+R 重开评审面板",
114
117
  },
115
118
  en: {
116
119
  close: "Esc close",
117
120
  scroll: "↑/↓ scroll",
118
121
  page: "PgUp/PgDn page",
119
122
  switchLane: "Tab & Shift + Tab switch lane",
123
+ reopen: "Ctrl+Shift+R reopen review overlay",
120
124
  },
121
125
  };
122
126
 
@@ -152,8 +152,12 @@ export interface ExecutionCheckpoint {
152
152
  * progress record. doneVcIds stays for legacy checkpoints and the final
153
153
  * audit pass. */
154
154
  tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
155
+ /** v0.7.1: watchdog round counter. Persisted so a session restart cannot
156
+ * silently hand a stalled run a fresh budget; optional so checkpoints
157
+ * written before this field keep loading. */
158
+ stallRounds?: number;
155
159
  /** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
156
- audit?: { rounds: number; lastResult?: string; passed?: boolean };
160
+ audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
157
161
  }
158
162
 
159
163
  export interface OwnerInfo {
@@ -461,7 +465,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
461
465
  const record = asRecord(value, label);
462
466
  rejectExtraKeys(
463
467
  record,
464
- new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "audit"]),
468
+ new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
465
469
  label,
466
470
  );
467
471
  const execution: ExecutionCheckpoint = {
@@ -505,14 +509,20 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
505
509
  }
506
510
  execution.tasks = tasks;
507
511
  }
512
+ if (record.stallRounds !== undefined && record.stallRounds !== null) {
513
+ execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
514
+ }
508
515
  if (record.audit !== undefined && record.audit !== null) {
509
516
  const audit = asRecord(record.audit, `${label}.audit`);
510
- rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed"]), `${label}.audit`);
511
- const parsed: { rounds: number; lastResult?: string; passed?: boolean } = {
517
+ rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
518
+ const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
512
519
  rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
513
520
  };
514
521
  if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
515
522
  if (audit.passed !== undefined) parsed.passed = asBool(audit.passed, `${label}.audit.passed`);
523
+ if (audit.undeterminable !== undefined) {
524
+ parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
525
+ }
516
526
  execution.audit = parsed;
517
527
  }
518
528
  return execution;
@@ -1071,7 +1081,9 @@ export interface ExecutionProgressInput {
1071
1081
  /** v0.6.1: task-tree progress snapshot (authoritative). */
1072
1082
  tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
1073
1083
  /** v0.6.1: completion-audit bookkeeping update. */
1074
- audit?: { rounds: number; lastResult?: string; passed?: boolean };
1084
+ audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
1085
+ /** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
1086
+ stallRounds?: number;
1075
1087
  }
1076
1088
 
1077
1089
  export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
@@ -1091,6 +1103,7 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
1091
1103
  if (progress.delegate === null) delete execution.delegate;
1092
1104
  else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
1093
1105
  if (progress.tasks !== undefined) execution.tasks = progress.tasks;
1106
+ if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
1094
1107
  if (progress.audit !== undefined) execution.audit = progress.audit;
1095
1108
  return { ...cp, execution };
1096
1109
  }
@@ -1,11 +1,17 @@
1
- /** Tests for the completion auditor (v0.6.1): verdict parsing, rollback
2
- * boundaries, skipped-pass, and no-cover exclusion. */
1
+ /** Tests for the execution reviewer: tri-state verdict parsing, contract
2
+ * coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
3
3
 
4
4
  import * as assert from "node:assert/strict";
5
5
  import { describe, it } from "node:test";
6
6
  import type { CheckItem } from "../src/plan.ts";
7
- import { buildTaskView } from "../src/tasks.ts";
8
- import { applyAuditOutcome, buildAuditTask, parseAuditReport, presolvedCheckIds } from "../src/auditor.ts";
7
+ import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
8
+ import {
9
+ applyAuditOutcome,
10
+ auditablePendingChecks,
11
+ buildAuditTask,
12
+ parseAuditReport,
13
+ presolvedCheckIds,
14
+ } from "../src/auditor.ts";
9
15
  import { parsePlanTasks } from "../src/plan.ts";
10
16
 
11
17
  const PLAN = `## Tasks
@@ -37,6 +43,8 @@ function view(progress?: Record<string, { status: "complete" | "skipped" }>) {
37
43
  return buildTaskView(parsePlanTasks(PLAN), progress);
38
44
  }
39
45
 
46
+ const ALL_IDS = ["VC-001", "VC-002", "VC-003", "VC-004"];
47
+
40
48
  describe("auditor parsing", () => {
41
49
  it("parses per-check verdicts and ignores unknown ids", () => {
42
50
  const report = [
@@ -44,48 +52,139 @@ describe("auditor parsing", () => {
44
52
  "- `VC-002` — verdict: fail; evidence: src/c.ts; note: missing",
45
53
  "- `VC-999` — verdict: pass; evidence: n/a",
46
54
  ].join("\n");
47
- const { passed, failed } = parseAuditReport(report, checks());
55
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
48
56
  assert.deepEqual(passed, ["VC-001"]);
49
57
  assert.deepEqual(failed, ["VC-002"]);
58
+ assert.deepEqual(undeterminable.sort(), ["VC-003", "VC-004"]);
59
+ });
60
+
61
+ it("parses verdicts wrapped in markdown emphasis", () => {
62
+ // Regression: `verdict: **pass**` used to parse as nothing, and the
63
+ // caller's fail-closed rule then rolled the entire run back.
64
+ const report = [
65
+ "- `VC-001` — verdict: **pass**; evidence: src/a.ts",
66
+ "- `VC-002` — verdict: **fail**; evidence: src/c.ts",
67
+ "- `VC-003` — verdict: *pass*; evidence: src/d.ts",
68
+ "- `VC-004` — verdict: `pass`; evidence: src/e.ts",
69
+ ].join("\n");
70
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
71
+ assert.deepEqual(passed, ["VC-001", "VC-003", "VC-004"]);
72
+ assert.deepEqual(failed, ["VC-002"]);
73
+ assert.deepEqual(undeterminable, []);
74
+ });
75
+
76
+ it("does not read a verdict word prefix as a verdict", () => {
77
+ const report = ["- `VC-001` — verdict: passed; evidence: x", "- `VC-002` — verdict: failing; evidence: y"].join("\n");
78
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
79
+ assert.deepEqual(passed, []);
80
+ assert.deepEqual(failed, []);
81
+ assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
50
82
  });
51
83
 
52
84
  it("conflicting verdicts for one check resolve to fail", () => {
53
85
  const report = "- `VC-001` — verdict: pass; note: a\n- `VC-001` — verdict: fail; note: b";
54
- const { passed, failed } = parseAuditReport(report, checks());
86
+ const { passed, failed } = parseAuditReport(report, ALL_IDS);
55
87
  assert.deepEqual(passed, []);
56
88
  assert.deepEqual(failed, ["VC-001"]);
57
89
  });
58
90
 
59
- it("builds the brief with the auditable checks only", () => {
91
+ it("reads an explicit undeterminable verdict", () => {
92
+ const report = "- `VC-001` — verdict: pass\n- `VC-002` — verdict: undeterminable; evidence: bash not granted";
93
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
94
+ assert.deepEqual(passed, ["VC-001"]);
95
+ assert.deepEqual(failed, []);
96
+ assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-003", "VC-004"]);
97
+ });
98
+
99
+ it("treats a reviewer-shaped report as undeterminable, never as failure", () => {
100
+ // Regression: agents/reviewer.md was the auditor's system prompt and
101
+ // mandates `## Findings` / `## Questions` with F-### lines and never
102
+ // says "verdict". Under fail-closed that whole shape read as "every
103
+ // check failed" and rolled the run back.
104
+ const report = [
105
+ "## Findings",
106
+ "",
107
+ "- `F-001` — severity: high; affected: VC-001; evidence: src/a.ts; impact: x; recommended fix: y; suggested disposition: accept.",
108
+ "",
109
+ "## Questions",
110
+ "",
111
+ "None.",
112
+ ].join("\n");
113
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
114
+ assert.deepEqual(passed, []);
115
+ assert.deepEqual(failed, [], "a reviewer-shaped report is not a failed audit");
116
+ assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
117
+ });
118
+
119
+ it("classifies a partially covered report per check", () => {
120
+ const report = "- `VC-001` — verdict: pass\n- `VC-003` — verdict: **pass**";
121
+ const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
122
+ assert.deepEqual(passed, ["VC-001", "VC-003"]);
123
+ assert.deepEqual(failed, []);
124
+ assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-004"]);
125
+ });
126
+
127
+ it("builds the brief with the auditable pending checks only", () => {
60
128
  const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
61
129
  assert.match(task, /read-only/i);
62
130
  assert.match(task, /`VC-001`/);
63
131
  assert.doesNotMatch(task, /`VC-999`/);
132
+ assert.doesNotMatch(task, /`VC-004`/, "a check covering no task is never audited");
133
+ });
134
+
135
+ it("brief states the tri-state contract and forbids fail-for-want-of-evidence", () => {
136
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
137
+ assert.match(task, /verdict: pass \| fail \| undeterminable/);
138
+ assert.match(task, /never report `fail` for want of evidence/i);
139
+ });
140
+
141
+ it("brief omits checks already done in an earlier round", () => {
142
+ // Regression: the brief listed every auditable check, so a round-2
143
+ // auditor that legitimately skipped the satisfied ones looked like a
144
+ // contract violation -- and, with no rollback to produce progress, the
145
+ // budget climbed to the cap and the run livelocked.
146
+ const done = buildAuditTask("/tmp/PLAN_v1.md", checks(["VC-001"]), view(), 2);
147
+ assert.doesNotMatch(done, /`VC-001`/);
148
+ assert.match(done, /`VC-002`/);
149
+ assert.deepEqual(
150
+ auditablePendingChecks(checks(["VC-001"]), view()).map((item) => item.id),
151
+ ["VC-002", "VC-003"],
152
+ );
153
+ });
154
+
155
+ it("applyAuditOutcome classifies without touching the checklist or task tree", () => {
156
+ const checklist = checks();
157
+ const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
158
+ const parsed = parseAuditReport("- `VC-001` — verdict: pass\n- `VC-002` — verdict: fail", ["VC-001", "VC-002"]);
159
+ const outcome = applyAuditOutcome(3, parsed, "report text");
160
+ assert.deepEqual(outcome.passed, ["VC-001"]);
161
+ assert.deepEqual(outcome.failed, ["VC-002"]);
162
+ assert.deepEqual(outcome.undeterminable, []);
163
+ assert.equal(outcome.round, 3);
164
+ assert.equal(outcome.report, "report text");
165
+ assert.equal(checklist.every((item) => item.done === false), true, "no done flag written here");
166
+ assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "complete", "no rollback here");
64
167
  });
65
168
  });
66
169
 
67
170
  describe("audit rollback boundaries", () => {
68
171
  it("failed multi-cover check rolls back every covered task", () => {
69
172
  const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
70
- const checklist = checks();
71
- const outcome = applyAuditOutcome(checklist, tasks, 1, ["VC-001"], ["VC-002"], "report");
72
- assert.deepEqual(outcome.rolledBack.sort(), ["Task-2", "Task-3", "Task-3.1"]);
73
- // Passed check stays done; its task stays closed.
74
- assert.equal(checks().length, 4);
173
+ const rolledBack = auditRollbackSet(tasks, checks(), "VC-002");
174
+ assert.deepEqual(rolledBack.sort(), ["Task-2", "Task-3", "Task-3.1"]);
175
+ // A check that passed keeps its task closed.
75
176
  assert.equal(tasks.find((t) => t.id === "Task-1")?.status, "complete");
76
177
  assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "pending");
77
178
  });
78
179
 
79
180
  it("covering the parent cascades to subtasks even when the parent alone is listed", () => {
80
181
  const tasks = view({ "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
81
- const outcome = applyAuditOutcome(checks(), tasks, 1, [], ["VC-002"], "report");
82
- assert.ok(outcome.rolledBack.includes("Task-3.1"), "subtask reopens with the parent");
182
+ assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-3.1"), "subtask reopens with the parent");
83
183
  });
84
184
 
85
185
  it("skipped tasks reopen too (skip state cleared)", () => {
86
186
  const tasks = view({ "Task-2": { status: "skipped" } });
87
- const outcome = applyAuditOutcome(checks(), tasks, 1, [], ["VC-002"], "report");
88
- assert.ok(outcome.rolledBack.includes("Task-2"));
187
+ assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-2"));
89
188
  assert.equal(tasks.find((t) => t.id === "Task-2")?.skipReason, undefined);
90
189
  });
91
190
 
@@ -108,4 +207,4 @@ describe("audit rollback boundaries", () => {
108
207
  assert.ok(presolved.includes("VC-003"));
109
208
  assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
110
209
  });
111
- });
210
+ });