pi-plans 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,6 +29,7 @@ import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from "
29
29
  import * as path from "node:path";
30
30
  import { StateError, atomicWriteJson, resolveStateRootOrNull, runGit, runDirPath, utcNow } from "./state.ts";
31
31
  import { assertOwnership, heldOwnershipRecord } from "./run-ownership.ts";
32
+ import type { NoProgressState, ReviewBudget } from "./review-budget.ts";
32
33
 
33
34
  // ---------------------------------------------------------------------------
34
35
  // Schema
@@ -133,6 +134,21 @@ export interface ExecutionApproval {
133
134
  approvedAt: string;
134
135
  }
135
136
 
137
+ /** v0.9.2: the newest failed execution-review round's rollback set plus the
138
+ * tasks among it that are still open. Captured when the round commits (the
139
+ * reopen helpers mutate the tree and report only flipped nodes, so a later
140
+ * re-derivation cannot rebuild this set) and persisted so wakes, the pause
141
+ * reason, the dashboard, and resume briefs can name the blocker after a
142
+ * restart. `escalatedRounds` counts consecutive blocked wakes — the escalation
143
+ * ladder must not ride `stallRounds`, which any successful tool call resets. */
144
+ export interface ExecutionBlocked {
145
+ rolledBack: string[];
146
+ tasks: string[];
147
+ round: number;
148
+ escalatedRounds: number;
149
+ since: string;
150
+ }
151
+
136
152
  export interface ExecutionCheckpoint {
137
153
  approval: ExecutionApproval | null;
138
154
  doneVcIds: string[];
@@ -156,8 +172,59 @@ export interface ExecutionCheckpoint {
156
172
  * silently hand a stalled run a fresh budget; optional so checkpoints
157
173
  * written before this field keep loading. */
158
174
  stallRounds?: number;
175
+ /** v0.9.2: outstanding blocker from the newest failed review round (see
176
+ * `ExecutionBlocked`); absent = nothing blocks the review. Optional on
177
+ * read so checkpoints written before it keep loading. */
178
+ blocked?: ExecutionBlocked | null;
179
+ /** v0.9.3: the per-run execution-review budget (committed rounds, or
180
+ * `"unlimited"`). Absent = not decided yet (the picker runs before round
181
+ * 1) OR written by a pre-v0.9.3 build, in which case a checkpoint that
182
+ * already spent rounds keeps the legacy 5-round bound
183
+ * (`resolveStoredBudget` in ./review-budget.ts). */
184
+ reviewBudget?: ReviewBudget;
185
+ /** v0.9.3: the budget above was the no-UI fallback, not a user pick — the
186
+ * dashboard/resume brief annotate it `(default)`. Optional on read. */
187
+ reviewBudgetDefaulted?: boolean;
188
+ /** v0.9.3: committed review rounds across the WHOLE run (never reset by a
189
+ * grant) — bounds the unlimited budget together with `reviewCapExtension`. */
190
+ reviewRoundsTotal?: number;
191
+ /** v0.9.3: explicit `/plans-execute` grants lifted the unlimited hard cap
192
+ * by this many rounds (each grant that lands on `unlimited` adds
193
+ * `UNLIMITED_HARD_CAP`). */
194
+ reviewCapExtension?: number;
195
+ /** v0.9.3: no-progress valve state for the unlimited budget (signature of
196
+ * the last committed round's outcome plus its consecutive streak). */
197
+ reviewNoProgress?: NoProgressState | null;
198
+ /** v0.9.1 (F-002): set when the execution-review loop mechanically
199
+ * appended finding tasks to the approved plan. The checkpoint's plan
200
+ * identity is re-stamped at that moment so /resume-plans accepts the
201
+ * amended plan; the approval record keeps the ORIGINAL digest as
202
+ * evidence of what the user actually approved. */
203
+ planAmended?: { sha256: string; amendedAt: string; round: number };
159
204
  /** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
160
- audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
205
+ audit?: {
206
+ rounds: number;
207
+ lastResult?: string;
208
+ passed?: boolean;
209
+ undeterminable?: string[];
210
+ /** v0.9: unresolved implementation findings from the newest committed
211
+ * review round (stable F-### ids). Optional so pre-v0.9 checkpoints
212
+ * keep loading as "no findings". */
213
+ findings?: ReviewFindingRecord[];
214
+ };
215
+ }
216
+
217
+ /** Serializable shape of one review finding (v0.9). Structurally identical to
218
+ * auditor.ts's ReviewFinding; declared here so the checkpoint layer does not
219
+ * import the auditor. */
220
+ export interface ReviewFindingRecord {
221
+ id: string;
222
+ severity: string;
223
+ taskIds: string[];
224
+ proposedTask?: string;
225
+ note: string;
226
+ evidence: string;
227
+ raw: string;
161
228
  }
162
229
 
163
230
  export interface OwnerInfo {
@@ -269,6 +336,13 @@ function asInt(value: unknown, label: string, min: number): number {
269
336
  return value;
270
337
  }
271
338
 
339
+ /** v0.9.3: a positive integer round count or the literal "unlimited". */
340
+ function asReviewBudget(value: unknown, label: string): ReviewBudget {
341
+ if (value === "unlimited") return "unlimited";
342
+ if (typeof value === "number" && Number.isSafeInteger(value) && value >= 1) return value;
343
+ throw new CheckpointValidationError(`${label}: expected a positive integer or "unlimited"`);
344
+ }
345
+
272
346
  function asEnum<T extends string>(value: unknown, allowed: Set<T>, label: string): T {
273
347
  const text = asString(value, label);
274
348
  if (!allowed.has(text as T)) {
@@ -465,7 +539,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
465
539
  const record = asRecord(value, label);
466
540
  rejectExtraKeys(
467
541
  record,
468
- new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
542
+ new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "blocked", "planAmended", "audit", "reviewBudget", "reviewBudgetDefaulted", "reviewRoundsTotal", "reviewCapExtension", "reviewNoProgress"]),
469
543
  label,
470
544
  );
471
545
  const execution: ExecutionCheckpoint = {
@@ -485,6 +559,17 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
485
559
  };
486
560
  if (record.currentI !== undefined) execution.currentI = asString(record.currentI, `${label}.currentI`);
487
561
  if (record.pausedReason !== undefined) execution.pausedReason = asString(record.pausedReason, `${label}.pausedReason`);
562
+ if (record.blocked !== undefined && record.blocked !== null) {
563
+ const blocked = asRecord(record.blocked, `${label}.blocked`);
564
+ rejectExtraKeys(blocked, new Set(["rolledBack", "tasks", "round", "escalatedRounds", "since"]), `${label}.blocked`);
565
+ execution.blocked = {
566
+ rolledBack: asStringArray(blocked.rolledBack, `${label}.blocked.rolledBack`),
567
+ tasks: asStringArray(blocked.tasks, `${label}.blocked.tasks`),
568
+ round: asInt(blocked.round, `${label}.blocked.round`, 0),
569
+ escalatedRounds: asInt(blocked.escalatedRounds, `${label}.blocked.escalatedRounds`, 0),
570
+ since: asTimestamp(blocked.since, `${label}.blocked.since`),
571
+ };
572
+ }
488
573
  if (record.reverifyAll !== undefined) execution.reverifyAll = asBool(record.reverifyAll, `${label}.reverifyAll`);
489
574
  if (record.originWorktree !== undefined) execution.originWorktree = asString(record.originWorktree, `${label}.originWorktree`);
490
575
  if (record.delegate !== undefined && record.delegate !== null) {
@@ -512,10 +597,43 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
512
597
  if (record.stallRounds !== undefined && record.stallRounds !== null) {
513
598
  execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
514
599
  }
600
+ if (record.planAmended !== undefined && record.planAmended !== null) {
601
+ const rec = asRecord(record.planAmended, `${label}.planAmended`);
602
+ rejectExtraKeys(rec, new Set(["sha256", "amendedAt", "round"]), `${label}.planAmended`);
603
+ execution.planAmended = {
604
+ sha256: asString(rec.sha256, `${label}.planAmended.sha256`),
605
+ amendedAt: asString(rec.amendedAt, `${label}.planAmended.amendedAt`),
606
+ round: asInt(rec.round, `${label}.planAmended.round`, 0),
607
+ };
608
+ }
609
+ // v0.9.3 review budget. `reviewBudget` accepts a positive integer or the
610
+ // literal "unlimited"; the counters are non-negative integers; the
611
+ // no-progress record is a small fixed shape. All optional on read so
612
+ // pre-feature checkpoints keep loading.
613
+ if (record.reviewBudget !== undefined && record.reviewBudget !== null) {
614
+ execution.reviewBudget = asReviewBudget(record.reviewBudget, `${label}.reviewBudget`);
615
+ }
616
+ if (record.reviewBudgetDefaulted !== undefined) {
617
+ execution.reviewBudgetDefaulted = asBool(record.reviewBudgetDefaulted, `${label}.reviewBudgetDefaulted`);
618
+ }
619
+ if (record.reviewRoundsTotal !== undefined && record.reviewRoundsTotal !== null) {
620
+ execution.reviewRoundsTotal = asInt(record.reviewRoundsTotal, `${label}.reviewRoundsTotal`, 0);
621
+ }
622
+ if (record.reviewCapExtension !== undefined && record.reviewCapExtension !== null) {
623
+ execution.reviewCapExtension = asInt(record.reviewCapExtension, `${label}.reviewCapExtension`, 0);
624
+ }
625
+ if (record.reviewNoProgress !== undefined && record.reviewNoProgress !== null) {
626
+ const np = asRecord(record.reviewNoProgress, `${label}.reviewNoProgress`);
627
+ rejectExtraKeys(np, new Set(["key", "streak"]), `${label}.reviewNoProgress`);
628
+ execution.reviewNoProgress = {
629
+ key: asString(np.key, `${label}.reviewNoProgress.key`),
630
+ streak: asInt(np.streak, `${label}.reviewNoProgress.streak`, 0),
631
+ };
632
+ }
515
633
  if (record.audit !== undefined && record.audit !== null) {
516
634
  const audit = asRecord(record.audit, `${label}.audit`);
517
- rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
518
- const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
635
+ rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable", "findings"]), `${label}.audit`);
636
+ const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] } = {
519
637
  rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
520
638
  };
521
639
  if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
@@ -523,6 +641,24 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
523
641
  if (audit.undeterminable !== undefined) {
524
642
  parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
525
643
  }
644
+ if (audit.findings !== undefined && audit.findings !== null) {
645
+ const arr = Array.isArray(audit.findings) ? audit.findings : null;
646
+ if (!arr) throw new CheckpointValidationError(`${label}.audit.findings must be an array`);
647
+ parsed.findings = arr.map((entry, i) => {
648
+ const rec = asRecord(entry, `${label}.audit.findings.${i}`);
649
+ rejectExtraKeys(rec, new Set(["id", "severity", "taskIds", "proposedTask", "note", "evidence", "raw"]), `${label}.audit.findings.${i}`);
650
+ const out: ReviewFindingRecord = {
651
+ id: asString(rec.id, `${label}.audit.findings.${i}.id`),
652
+ severity: asString(rec.severity, `${label}.audit.findings.${i}.severity`),
653
+ taskIds: rec.taskIds === undefined ? [] : asStringArray(rec.taskIds, `${label}.audit.findings.${i}.taskIds`),
654
+ note: rec.note === undefined ? "" : asString(rec.note, `${label}.audit.findings.${i}.note`),
655
+ evidence: rec.evidence === undefined ? "" : asString(rec.evidence, `${label}.audit.findings.${i}.evidence`),
656
+ raw: rec.raw === undefined ? "" : asString(rec.raw, `${label}.audit.findings.${i}.raw`),
657
+ };
658
+ if (rec.proposedTask !== undefined) out.proposedTask = asString(rec.proposedTask, `${label}.audit.findings.${i}.proposedTask`);
659
+ return out;
660
+ });
661
+ }
526
662
  execution.audit = parsed;
527
663
  }
528
664
  return execution;
@@ -1080,10 +1216,26 @@ export interface ExecutionProgressInput {
1080
1216
  delegate?: { modelSelector: string; startedAt: string } | null;
1081
1217
  /** v0.6.1: task-tree progress snapshot (authoritative). */
1082
1218
  tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
1083
- /** v0.6.1: completion-audit bookkeeping update. */
1084
- audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] };
1219
+ /** v0.6.1: completion-audit bookkeeping update. v0.9: findings rides the
1220
+ * same replace-semantics slot — callers that must preserve findings (e.g.
1221
+ * budget renewal) pass them through explicitly. */
1222
+ audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] };
1085
1223
  /** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
1086
1224
  stallRounds?: number;
1225
+ /** v0.9.2: blocker record; null clears it (delete-on-null, like
1226
+ * `pausedReason`/`delegate`). */
1227
+ blocked?: ExecutionBlocked | null;
1228
+ /** v0.9.3: per-run execution-review budget; null clears it (delete-on-null,
1229
+ * used by the stop/complete cleanup paths). */
1230
+ reviewBudget?: ReviewBudget | null;
1231
+ /** v0.9.3: whether `reviewBudget` came from the no-UI fallback. */
1232
+ reviewBudgetDefaulted?: boolean | null;
1233
+ /** v0.9.3: committed rounds across the whole run (monotonic). */
1234
+ reviewRoundsTotal?: number;
1235
+ /** v0.9.3: explicit unlimited-cap extension accumulated by grants. */
1236
+ reviewCapExtension?: number;
1237
+ /** v0.9.3: no-progress valve state; null clears it (delete-on-null). */
1238
+ reviewNoProgress?: NoProgressState | null;
1087
1239
  }
1088
1240
 
1089
1241
  export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: ExecutionProgressInput): WorkflowCheckpoint {
@@ -1104,10 +1256,37 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
1104
1256
  else if (progress.delegate !== undefined) execution.delegate = progress.delegate;
1105
1257
  if (progress.tasks !== undefined) execution.tasks = progress.tasks;
1106
1258
  if (progress.stallRounds !== undefined) execution.stallRounds = progress.stallRounds;
1259
+ if (progress.blocked === null) delete execution.blocked;
1260
+ else if (progress.blocked !== undefined) execution.blocked = progress.blocked;
1261
+ if (progress.reviewBudget === null) delete execution.reviewBudget;
1262
+ else if (progress.reviewBudget !== undefined) execution.reviewBudget = progress.reviewBudget;
1263
+ if (progress.reviewBudgetDefaulted === null) delete execution.reviewBudgetDefaulted;
1264
+ else if (progress.reviewBudgetDefaulted !== undefined) execution.reviewBudgetDefaulted = progress.reviewBudgetDefaulted;
1265
+ if (progress.reviewRoundsTotal !== undefined) execution.reviewRoundsTotal = progress.reviewRoundsTotal;
1266
+ if (progress.reviewCapExtension !== undefined) execution.reviewCapExtension = progress.reviewCapExtension;
1267
+ if (progress.reviewNoProgress === null) delete execution.reviewNoProgress;
1268
+ else if (progress.reviewNoProgress !== undefined) execution.reviewNoProgress = progress.reviewNoProgress;
1107
1269
  if (progress.audit !== undefined) execution.audit = progress.audit;
1108
1270
  return { ...cp, execution };
1109
1271
  }
1110
1272
 
1273
+ /** v0.9.1 (F-002): the execution-review loop appended finding tasks to the
1274
+ * approved plan file; re-stamp the checkpoint's plan identity to the amended
1275
+ * digest so a later /resume-plans does not reject the run as plan-mismatch.
1276
+ * The approval record is untouched — it keeps the digest the user actually
1277
+ * approved, and planAmended records when and why the identity moved. */
1278
+ export function applyExecutionPlanAmended(cp: WorkflowCheckpoint, plan: PlanIdentity, round: number): WorkflowCheckpoint {
1279
+ if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1280
+ return {
1281
+ ...cp,
1282
+ plan,
1283
+ execution: {
1284
+ ...cp.execution,
1285
+ planAmended: { sha256: plan.sha256, amendedAt: utcNow(), round },
1286
+ },
1287
+ };
1288
+ }
1289
+
1111
1290
  /** D-011/F-001: code state changed under an unchanged plan — keep authorization, re-verify first. */
1112
1291
  export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheckpoint {
1113
1292
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
@@ -1118,14 +1297,25 @@ export function applyExecutionCompleted(cp: WorkflowCheckpoint): WorkflowCheckpo
1118
1297
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1119
1298
  // v0.6.1 (D-018): the post-execution amelioration loop is gone; a
1120
1299
  // completed audit passes the run straight to the terminal phase.
1300
+ // v0.9.3: the review budget (and its valve state) describes a review that
1301
+ // no longer runs — completed runs do not carry it.
1302
+ const {
1303
+ reviewBudget: _reviewBudget,
1304
+ reviewBudgetDefaulted: _reviewBudgetDefaulted,
1305
+ reviewRoundsTotal: _reviewRoundsTotal,
1306
+ reviewCapExtension: _reviewCapExtension,
1307
+ reviewNoProgress: _reviewNoProgress,
1308
+ ...execution
1309
+ } = cp.execution;
1121
1310
  return {
1122
1311
  ...cp,
1123
1312
  phase: "completed",
1124
1313
  nextAction: "none",
1125
1314
  execution: {
1126
- ...cp.execution,
1315
+ ...execution,
1127
1316
  pausedReason: undefined,
1128
1317
  delegate: undefined,
1318
+ blocked: undefined,
1129
1319
  audit: { ...(cp.execution.audit ?? { rounds: 0 }), passed: true },
1130
1320
  },
1131
1321
  };
@@ -1156,6 +1346,14 @@ export function applyMigration(
1156
1346
  // migration (same rule as VC validity).
1157
1347
  tasks: {},
1158
1348
  audit: { rounds: 0 },
1349
+ // v0.9.3 (F-005): the review budget and its counters ARE carried
1350
+ // across a migration — they describe the user's choice and the
1351
+ // run's cumulative spend, not the (invalidated) code state.
1352
+ reviewBudget: cp.execution.reviewBudget,
1353
+ reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted,
1354
+ reviewRoundsTotal: cp.execution.reviewRoundsTotal,
1355
+ reviewCapExtension: cp.execution.reviewCapExtension,
1356
+ reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
1159
1357
  }
1160
1358
  : undefined,
1161
1359
  implementationReview: cp.implementationReview
@@ -1179,7 +1377,18 @@ function migrationNextAction(cp: WorkflowCheckpoint): NextAction {
1179
1377
  /** Mark a paused stop without erasing the last phase (D-008). */
1180
1378
  export function applyExecutionStopped(cp: WorkflowCheckpoint, reason: string): WorkflowCheckpoint {
1181
1379
  if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
1182
- const { delegate: _delegate, ...execution } = cp.execution;
1380
+ // v0.9.3: a stop also drops the review budget and its valve state — the
1381
+ // next execution of this plan picks its own budget.
1382
+ const {
1383
+ delegate: _delegate,
1384
+ blocked: _blocked,
1385
+ reviewBudget: _reviewBudget,
1386
+ reviewBudgetDefaulted: _reviewBudgetDefaulted,
1387
+ reviewRoundsTotal: _reviewRoundsTotal,
1388
+ reviewCapExtension: _reviewCapExtension,
1389
+ reviewNoProgress: _reviewNoProgress,
1390
+ ...execution
1391
+ } = cp.execution;
1183
1392
  // A stop also revokes any outstanding audit-rollback authorization.
1184
1393
  const audit = execution.audit ? { ...execution.audit } : undefined;
1185
1394
  return { ...cp, execution: { ...execution, audit, pausedReason: reason } };
@@ -303,7 +303,7 @@ describe("analyze_refs fanout", () => {
303
303
 
304
304
  it("pins the per-batch overlay lifecycle (open before spawn, close in finally, cap 3)", () => {
305
305
  const source = fs.readFileSync(path.join(ROOT, "tools", "analyze-refs.ts"), "utf8");
306
- assert.equal((source.match(/new RefineOverlayController\("refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
306
+ assert.equal((source.match(/new RefineOverlayController\(\s*"refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
307
307
  assert.match(source, /overlay\?\.open\(refineOverlayContext\(ctx\), modelLabel\)/);
308
308
  assert.match(source, /await overlay\?\.close\(\);/);
309
309
  assert.match(source, /const BATCH_SIZE = 3;/);
@@ -2,6 +2,9 @@
2
2
  * coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
3
3
 
4
4
  import * as assert from "node:assert/strict";
5
+ import * as fs from "node:fs";
6
+ import * as os from "node:os";
7
+ import * as path from "node:path";
5
8
  import { describe, it } from "node:test";
6
9
  import type { CheckItem } from "../src/plan.ts";
7
10
  import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
@@ -207,4 +210,185 @@ describe("audit rollback boundaries", () => {
207
210
  assert.ok(presolved.includes("VC-003"));
208
211
  assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
209
212
  });
210
- });
213
+ });
214
+ describe("findings parsing (v0.9)", () => {
215
+ const GOOD = [
216
+ "- `VC-001` — verdict: pass; evidence: ok",
217
+ "",
218
+ "- `F-001` — severity: high; tasks: Task-3, task-4; note: union rollback missing; evidence: src/exec.ts:1290",
219
+ "- `F-002` — severity: medium; tasks: none; proposed-task: cap retry backoff at 60s; note: unbounded; evidence: src/client.ts:12",
220
+ ].join("\n");
221
+
222
+ it("parses well-formed findings with id/task normalization", () => {
223
+ const { findings } = parseAuditReport(GOOD, ALL_IDS);
224
+ assert.equal(findings.length, 2);
225
+ const f1 = findings[0];
226
+ assert.equal(f1.id, "F-001");
227
+ assert.equal(f1.severity, "high");
228
+ assert.deepEqual(f1.taskIds, ["Task-3", "Task-4"]);
229
+ assert.equal(f1.note, "union rollback missing");
230
+ assert.equal(f1.evidence, "src/exec.ts:1290");
231
+ const f2 = findings[1];
232
+ assert.equal(f2.severity, "medium");
233
+ assert.deepEqual(f2.taskIds, []);
234
+ assert.equal(f2.proposedTask, "cap retry backoff at 60s");
235
+ });
236
+
237
+ it("tolerates emphasis markers on severity", () => {
238
+ const { findings } = parseAuditReport("- `F-001` — severity: **high**; tasks: Task-1; note: n; evidence: e", ALL_IDS);
239
+ assert.equal(findings[0]?.severity, "high");
240
+ const { findings: f2 } = parseAuditReport("- `F-002` — severity: `medium`; tasks: none; note: n", ALL_IDS);
241
+ assert.equal(f2[0]?.severity, "medium");
242
+ });
243
+
244
+ it("degrades unreadable severity to a recorded non-blocking entry (never a rollback driver)", () => {
245
+ const { findings } = parseAuditReport("- `F-003` — tasks: whatever; note: no severity field", ALL_IDS);
246
+ assert.equal(findings[0]?.severity, "malformed");
247
+ assert.deepEqual(findings[0]?.taskIds, []);
248
+ });
249
+
250
+ it("degrades a missing mandatory tasks field", () => {
251
+ const { findings } = parseAuditReport("- `F-004` — severity: high; note: tasks field absent", ALL_IDS);
252
+ assert.equal(findings[0]?.severity, "malformed");
253
+ });
254
+
255
+ it("resolves duplicate ids to the first bullet", () => {
256
+ const report = [
257
+ "- `F-001` — severity: high; tasks: Task-1; note: first",
258
+ "- `F-001` — severity: low; tasks: none; note: second",
259
+ ].join("\n");
260
+ const { findings } = parseAuditReport(report, ALL_IDS);
261
+ assert.equal(findings.length, 1);
262
+ assert.equal(findings[0].note, "first");
263
+ });
264
+
265
+ it("a finding bullet citing a VC verdict never registers that verdict", () => {
266
+ const report = "- `F-005` — severity: high; tasks: Task-1; note: cites VC-004 verdict: pass; evidence: z";
267
+ const { passed, failed } = parseAuditReport(report, ALL_IDS);
268
+ assert.equal(passed.length + failed.length, 0);
269
+ });
270
+
271
+ it("prose mentioning F-### outside a bullet is ignored", () => {
272
+ const { findings } = parseAuditReport("also prose mentions F-009 not a bullet\n- `F-001` — severity: low; tasks: none; note: real", ALL_IDS);
273
+ assert.deepEqual(findings.map((f) => f.id), ["F-001"]);
274
+ });
275
+
276
+ it("applyAuditOutcome carries findings through to the outcome", () => {
277
+ const parsed = parseAuditReport(GOOD, ["VC-001"]);
278
+ const outcome = applyAuditOutcome(1, parsed, GOOD);
279
+ assert.equal(outcome.findings?.length, 2);
280
+ });
281
+ });
282
+
283
+ describe("section-aware verdict parsing (v0.9.1 F-011)", () => {
284
+ it("parses heading-style sections: id heading + verdict on its own line below", () => {
285
+ const report = [
286
+ "## 1. Verification verdicts",
287
+ "",
288
+ "### VC-001",
289
+ "- verdict: pass; evidence: src/a.ts",
290
+ "",
291
+ "### VC-002",
292
+ "- verdict: **fail**; evidence: src/c.ts",
293
+ "",
294
+ "### VC-003",
295
+ "verdict: undeterminable",
296
+ ].join("\n");
297
+ const { passed, failed, undeterminable } = parseAuditReport(report, ["VC-001", "VC-002", "VC-003"]);
298
+ assert.deepEqual(passed, ["VC-001"]);
299
+ assert.deepEqual(failed, ["VC-002"]);
300
+ assert.deepEqual(undeterminable, ["VC-003"]);
301
+ });
302
+
303
+ it("a bare verdict with no open section is ignored (never misattributed)", () => {
304
+ const report = "Some preamble mentioning verdict: pass with no section above\n### VC-001\n- verdict: fail";
305
+ const { passed, failed } = parseAuditReport(report, ["VC-001"]);
306
+ assert.deepEqual(passed, []);
307
+ assert.deepEqual(failed, ["VC-001"]);
308
+ });
309
+
310
+ it("an unknown section id does not capture later bare verdicts", () => {
311
+ const report = ["### VC-999", "- verdict: pass", "### VC-002", "- verdict: pass"].join("\n");
312
+ const { passed, undeterminable } = parseAuditReport(report, ["VC-002"]);
313
+ assert.deepEqual(passed, ["VC-002"]);
314
+ assert.deepEqual(undeterminable, []);
315
+ });
316
+
317
+ it("finding bullets never open a verdict section", () => {
318
+ const report = [
319
+ "### VC-001",
320
+ "- `F-005` — severity: high; tasks: Task-1; note: cites verdict: pass inside a note; evidence: e",
321
+ "- verdict: fail",
322
+ ].join("\n");
323
+ const { passed, failed } = parseAuditReport(report, ["VC-001"]);
324
+ assert.deepEqual(passed, []);
325
+ assert.deepEqual(failed, ["VC-001"]);
326
+ assert.equal(parseAuditReport(report, ["VC-001"]).findings[0]?.severity, "high");
327
+ });
328
+
329
+ it("knownTaskIds filters finding mappings to plan tasks (F-003)", () => {
330
+ const report = "- `F-001` — severity: high; tasks: Task-1, VC-007, bogus; note: n; evidence: e";
331
+ const { findings } = parseAuditReport(report, [], new Set(["Task-1"]));
332
+ assert.deepEqual(findings[0]?.taskIds, ["Task-1"]);
333
+ });
334
+ });
335
+
336
+ describe("review brief dual-output contract (v0.9)", () => {
337
+ it("demands both sections: verdicts and findings grammar", () => {
338
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
339
+ assert.match(task, /1\. Verification verdicts/);
340
+ assert.match(task, /2\. Implementation findings/);
341
+ assert.match(task, /severity: high \| medium \| low/);
342
+ assert.match(task, /proposed-task:/);
343
+ });
344
+
345
+ it("lists plan tasks as the valid mapping domain", () => {
346
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
347
+ assert.match(task, /Plan tasks \(the only ids valid in a finding's tasks field\):/);
348
+ assert.match(task, /`Task-3\.1`: injection/);
349
+ });
350
+
351
+ it("injects prior unresolved findings with the stable-id reuse instruction", () => {
352
+ const prior = [{
353
+ id: "F-001", severity: "high" as const, taskIds: ["Task-2"], note: "still broken",
354
+ evidence: "src/b.ts", raw: "- `F-001` — severity: high; tasks: Task-2; note: still broken",
355
+ }];
356
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 3, prior);
357
+ assert.match(task, /reuse these exact ids while the problem persists/);
358
+ assert.match(task, /`F-001` — severity: high; tasks: Task-2/);
359
+ });
360
+
361
+ it("marks the first findings round when no prior list exists", () => {
362
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
363
+ assert.match(task, /first round with findings in scope/);
364
+ });
365
+ });
366
+
367
+ describe("review round report findings lines (v0.9)", () => {
368
+ it("emits the high-findings line and per-finding detail", async () => {
369
+ const { writeReviewRoundReport } = await import("../src/auditor.ts");
370
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-round-report-"));
371
+ try {
372
+ const file = writeReviewRoundReport(dir, {
373
+ budgetRound: 2,
374
+ attempt: 1,
375
+ outcome: "failed",
376
+ passed: ["VC-001"],
377
+ failed: [],
378
+ undeterminable: [],
379
+ findings: [
380
+ { id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "e", raw: "raw" },
381
+ { id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy up", note: "polish", evidence: "e", raw: "raw" },
382
+ ],
383
+ coveredTaskIds: ["Task-1", "Task-2"],
384
+ report: "## Report\nbody",
385
+ });
386
+ assert.ok(file);
387
+ const text = fs.readFileSync(file, "utf8");
388
+ assert.match(text, /- high findings: F-001/);
389
+ assert.match(text, /F-002 \(medium; proposed: tidy up\)/);
390
+ } finally {
391
+ fs.rmSync(dir, { recursive: true, force: true });
392
+ }
393
+ });
394
+ });