pi-plans 0.8.0 → 0.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/exec.ts CHANGED
@@ -57,6 +57,7 @@ import {
57
57
  applyExecutionProgress,
58
58
  applyExecutionStopped,
59
59
  createCheckpoint,
60
+ applyExecutionPlanAmended,
60
61
  loadCheckpoint,
61
62
  mutateCheckpoint,
62
63
  planIdentityOf,
@@ -72,6 +73,7 @@ import { resolveGraphMode } from "./code-graph/mode.ts";
72
73
  import {
73
74
  parseChecklist,
74
75
  parsePlanTasks,
76
+ CHECKLIST_HEADERS,
75
77
  flattenTasks,
76
78
  type CheckItem,
77
79
  type PlanTasks,
@@ -82,8 +84,10 @@ import {
82
84
  auditableChecks,
83
85
  buildTaskView,
84
86
  currentTask,
87
+ findingsRollbackSet,
85
88
  flattenTaskViews,
86
89
  invalidateChecksForRolledBackTasks,
90
+ maxWave,
87
91
  taskIsTerminal,
88
92
  taskProgress,
89
93
  taskProgressMap,
@@ -99,13 +103,15 @@ import {
99
103
  renderDashboardLines,
100
104
  renderDashboardTreeLines,
101
105
  } from "./dashboard.ts";
102
- import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditRoundResult } from "./auditor.ts";
106
+ import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
107
+ import type { ReviewFindingRecord } from "./workflow-state.ts";
103
108
  import { staleReloadHint as probeStaleReload } from "./staleness.ts";
104
109
  import { messaging } from "./messaging.ts";
105
110
  import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
106
111
  import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
107
112
  import { resolveReviewerSpawn } from "./thinking-levels.ts";
108
113
  import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
114
+ import { matchesTerminalKey } from "./terminal-keys.ts";
109
115
 
110
116
  export interface ExecState {
111
117
  planPath: string;
@@ -127,7 +133,7 @@ export interface ExecState {
127
133
  /** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
128
134
  * only when a round outcome commits (never on discard/cancel) — and is the
129
135
  * only piece persisted. */
130
- audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
136
+ audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean; findings: ReviewFinding[] };
131
137
  /** Execution-review loop (v0.8), memory-only: the attempt index names the
132
138
  * per-round report files; consecutiveDiscards bounds the fingerprint
133
139
  * re-run loop; inFlight owns the round's abort lifecycle. */
@@ -229,6 +235,10 @@ export interface CheckpointExecutionLoad {
229
235
  /** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
230
236
  * detected — resume requires a fresh handoff approval. */
231
237
  legacyDelegate?: boolean;
238
+ /** v0.9.1 (F-005): unresolved findings from the newest committed round,
239
+ * so the /resume-plans brief can surface outstanding highs before the
240
+ * per-turn injection ever runs. */
241
+ findings?: ReviewFinding[];
232
242
  error?: string;
233
243
  }
234
244
 
@@ -314,6 +324,7 @@ export function loadExecutionFromCheckpoint(
314
324
  rounds: cp.execution.audit?.rounds ?? 0,
315
325
  failed: [],
316
326
  undeterminable: cp.execution.audit?.undeterminable ?? [],
327
+ findings: toReviewFindings(cp.execution.audit?.findings),
317
328
  running: false,
318
329
  },
319
330
  review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
@@ -362,6 +373,7 @@ export function loadExecutionFromCheckpoint(
362
373
  pausedReason: cp.execution.pausedReason,
363
374
  legacyPlan: planTasks.legacy,
364
375
  legacyDelegate,
376
+ findings: toReviewFindings(cp.execution.audit?.findings),
365
377
  };
366
378
  }
367
379
 
@@ -501,6 +513,7 @@ function updatePanelWidget(ctx: ExtensionContext): void {
501
513
  auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
502
514
  auditFailed: current.audit.failed,
503
515
  auditUndeterminable: current.audit.undeterminable,
516
+ findings: current.audit.findings,
504
517
  reviewRunning: current.audit.running === true || current.review.inFlight !== null,
505
518
  startedAt: current.startedAt,
506
519
  usage: current.usage,
@@ -525,6 +538,7 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
525
538
  auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
526
539
  auditFailed: execution.audit.failed,
527
540
  auditUndeterminable: execution.audit.undeterminable,
541
+ findings: execution.audit.findings,
528
542
  reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
529
543
  });
530
544
  ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
@@ -597,7 +611,7 @@ function persist(ctx: ExtensionContext): void {
597
611
  startedAt: execution.startedAt,
598
612
  usage: execution.usage,
599
613
  stall: execution.stall,
600
- audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
614
+ audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
601
615
  });
602
616
  }
603
617
 
@@ -640,7 +654,7 @@ export async function startExecution(
640
654
  usage: { inToks: 0, outToks: 0 },
641
655
  uiLanguage: resolveUiLanguage(ctx.cwd),
642
656
  stall: { rounds: 0, lastSnapshot: null, paused: false },
643
- audit: { rounds: 0, failed: [], undeterminable: [], running: false },
657
+ audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
644
658
  review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
645
659
  auditLatch: { auditedThisSettle: false, activity: 0 },
646
660
  };
@@ -711,6 +725,9 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
711
725
  rounds: execution!.audit.rounds,
712
726
  lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
713
727
  undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
728
+ // v0.9: audit writes are replace-semantics — every writer must
729
+ // carry findings or a task update would silently wipe them.
730
+ findings: execution!.audit.findings,
714
731
  },
715
732
  }),
716
733
  );
@@ -737,10 +754,10 @@ export function recordExecutionTurn(
737
754
  }
738
755
 
739
756
  /** Test hook: replace the audit subagent with a deterministic function. */
740
- let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
757
+ let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
741
758
 
742
759
  export function __setAuditRunnerForTests(
743
- runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
760
+ runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
744
761
  ): void {
745
762
  auditRunnerForTests = runner;
746
763
  }
@@ -894,8 +911,30 @@ function abortInFlightReview(): void {
894
911
  activeReviewChain = null;
895
912
  }
896
913
 
914
+ /** v0.9: unresolved high-severity findings from the newest committed round.
915
+ * Presence in the newest round's report IS the unresolved set (stable ids:
916
+ * a problem is resolved only by no longer being reported). */
917
+ function unresolvedHighFindings(ex: ExecState): ReviewFinding[] {
918
+ return ex.audit.findings.filter((f) => f.severity === "high");
919
+ }
920
+
921
+ /** v0.9: checkpoint records -> runtime findings. Invalid severities degrade to
922
+ * "malformed" (recorded, non-blocking) — the same never-fail posture as the
923
+ * report parser. */
924
+ function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
925
+ if (!records) return [];
926
+ return records.map((r) => {
927
+ const severity = (r.severity === "high" || r.severity === "medium" || r.severity === "low") ? r.severity : "malformed";
928
+ return { id: r.id, severity, taskIds: Array.isArray(r.taskIds) ? r.taskIds : [], proposedTask: r.proposedTask, note: r.note ?? "", evidence: r.evidence ?? "", raw: r.raw ?? "" };
929
+ });
930
+ }
931
+
932
+ /** v0.9: findings widen the owed predicate — an unresolved high finding keeps
933
+ * the review owed even when every check is done (the stranded-high path),
934
+ * mirroring how a failed check keeps it owed today. */
897
935
  function reviewOwed(ex: ExecState): boolean {
898
- return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
936
+ return allTasksTerminal(ex.tasks)
937
+ && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
899
938
  }
900
939
 
901
940
  function runDirOf(ctx: ExtensionContext): string | null {
@@ -996,11 +1035,21 @@ function freshReviewLane(attempt: number): RefineLaneState {
996
1035
 
997
1036
  /** Fresh controller per round (the controller is one-shot: closed latch,
998
1037
  * overlayPromise bail, terminal-lane early return — reuse drops progress).
999
- * A UI failure must NEVER kill the round itself — best-effort only. */
1038
+ * A UI failure must NEVER kill the round itself — best-effort only. The
1039
+ * overlay forwards unhandled keys so Ctrl+Shift+T keeps working while the
1040
+ * review overlay holds focus (pi-tui has no key bubbling). */
1000
1041
  function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
1001
1042
  if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
1002
1043
  try {
1003
- const controller = new RefineOverlayController("auditor", [{ id: lane.id, label: lane.label }], () => {}, lang ?? "en");
1044
+ const controller = new RefineOverlayController(
1045
+ "auditor",
1046
+ [{ id: lane.id, label: lane.label }],
1047
+ () => {},
1048
+ lang ?? "en",
1049
+ (data) => {
1050
+ if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
1051
+ },
1052
+ );
1004
1053
  controller.seedLane(lane);
1005
1054
  controller.open(refineOverlayContext(ctx), modelLabel);
1006
1055
  return controller;
@@ -1013,15 +1062,20 @@ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "
1013
1062
  * state; inert when no round is in flight. */
1014
1063
  export function reopenReviewOverlay(ctx: ExtensionContext): void {
1015
1064
  if (!execution?.review.inFlight || !reviewLane) return;
1065
+ // Never stack a second overlay on a live one (Ctrl+Shift+R while open).
1066
+ if (reviewOverlay && !reviewOverlay.isClosed()) return;
1016
1067
  const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
1017
1068
  if (controller) reviewOverlay = controller;
1018
1069
  }
1019
1070
 
1020
1071
  function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
1072
+ const highs = unresolvedHighFindings(ex);
1021
1073
  const detail =
1022
1074
  ex.audit.failed.length > 0
1023
- ? `failed: ${ex.audit.failed.join(", ")}`
1024
- : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1075
+ ? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
1076
+ : highs.length > 0
1077
+ ? `high findings: ${highs.map((f) => f.id).join(", ")}`
1078
+ : `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
1025
1079
  const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
1026
1080
  pauseForStall(ctx, reason);
1027
1081
  // In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
@@ -1044,7 +1098,10 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1044
1098
  // Only auditable checks (with task coverage) gate completion; checks that
1045
1099
  // cover no task can never be verified and never block or complete.
1046
1100
  const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
1047
- if (pendingChecks.length === 0) {
1101
+ // v0.9: unresolved high findings block the fast completion path — they
1102
+ // keep the review owed instead (liveness: a high can never be completed
1103
+ // around, only fixed or paused at the cap).
1104
+ if (pendingChecks.length === 0 && unresolvedHighFindings(ex).length === 0) {
1048
1105
  await completeExecution(ctx);
1049
1106
  return;
1050
1107
  }
@@ -1075,12 +1132,13 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
1075
1132
  let result: AuditRoundResult;
1076
1133
  try {
1077
1134
  result = auditRunnerForTests
1078
- ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
1135
+ ? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt, priorFindings: ex.audit.findings })
1079
1136
  : await runCompletionAudit(ctx, {
1080
1137
  planPath: ex.planPath,
1081
1138
  checklist: ex.items,
1082
1139
  tasks: ex.tasks,
1083
1140
  round: round.attempt,
1141
+ priorFindings: ex.audit.findings,
1084
1142
  model: spawn.model,
1085
1143
  thinkingLevel: spawn.thinkingLevel,
1086
1144
  timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
@@ -1160,6 +1218,9 @@ async function handleReviewOutcome(
1160
1218
  passed: [],
1161
1219
  failed: [],
1162
1220
  undeterminable: pendingIds,
1221
+ // v0.9.1 (F-014): the discard report must not understate a
1222
+ // round whose embedded report carries highs.
1223
+ findings: owner.audit.findings,
1163
1224
  discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
1164
1225
  fingerprintCaptured: round.fingerprint,
1165
1226
  fingerprintFound: fingerprintNow,
@@ -1187,11 +1248,73 @@ async function handleReviewOutcome(
1187
1248
  await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
1188
1249
  }
1189
1250
 
1251
+ /** v0.9 (findings-driven fix loop): append plan tasks for high findings no
1252
+ * existing task owns. The reviewer stays read-only — it proposes the title
1253
+ * (proposed-task); this machinery applies it with provenance, so every high
1254
+ * finding always has an owner the executor can close, and the stable F-###
1255
+ * id rides in the task title for traceability. Best-effort: an unwritable
1256
+ * plan must not crash the loop (the finding then stays stranded and the cap
1257
+ * pause surfaces it). Returns the appended task ids. */
1258
+ function appendFindingTasks(ex: ExecState, highs: ReviewFinding[]): string[] {
1259
+ const appended: string[] = [];
1260
+ let next = flattenTaskViews(ex.tasks).reduce((max, t) => {
1261
+ const m = /^Task-(\d+)$/.exec(t.id);
1262
+ return m ? Math.max(max, Number(m[1])) : max;
1263
+ }, 0);
1264
+ const wave = maxWave(ex.tasks) + 1;
1265
+ try {
1266
+ const planText = fs.readFileSync(ex.planPath, "utf8");
1267
+ const lines = planText.split("\n");
1268
+ // Insert inside the Tasks section: before the Execution Waves
1269
+ // subsection when present, else before the FIRST checklist header the
1270
+ // plan uses (v0.9.1 F-013: legacy plans say `## Verifier Checklist`,
1271
+ // and an EOF fallback would land outside every parsed section), else
1272
+ // at EOF; walk back over blank separators so the bullet lands
1273
+ // adjacent to its siblings.
1274
+ let insertAt = lines.length;
1275
+ const wavesIdx = lines.findIndex((l) => /^###\s+Execution Waves/.test(l));
1276
+ const checklistIdx = lines.findIndex((l) => CHECKLIST_HEADERS.some((h) => new RegExp(`^##\\s+${h}`).test(l)));
1277
+ if (wavesIdx !== -1) insertAt = wavesIdx;
1278
+ else if (checklistIdx !== -1) insertAt = checklistIdx;
1279
+ while (insertAt > 0 && lines[insertAt - 1].trim() === "") insertAt--;
1280
+ // v0.9.1 (F-012): reviewer text becomes task-title metadata at parse
1281
+ // time — strip the microsyntax metacharacters (em/en dashes, `--`
1282
+ // separators, the `;` field delimiter) so an embedded token can never
1283
+ // split title from tail or forge fields.
1284
+ const sanitize = (text: string): string => text.replace(/[—–]/g, "-").replace(/-{2,}/g, "-").replace(/;/g, ",");
1285
+ const entries: Array<{ id: string; title: string }> = [];
1286
+ const newLines: string[] = [];
1287
+ for (const h of highs) {
1288
+ next += 1;
1289
+ const id = `Task-${next}`;
1290
+ const title = `fix ${h.id}: ${sanitize(h.proposedTask ?? h.note ?? "address the finding")} (appended by execution review round ${ex.audit.rounds})`;
1291
+ // v0.9.1 (F-006): carry the wave in the bullet tail so a re-parse
1292
+ // restores the same wave the live tree assigned — without it the
1293
+ // appended remediation task fell back to wave 1 on restore and
1294
+ // hijacked the ▸ anchor.
1295
+ newLines.push(`- \`${id}\`: ${title} — wave: ${wave}`);
1296
+ entries.push({ id, title });
1297
+ }
1298
+ lines.splice(insertAt, 0, ...newLines);
1299
+ fs.writeFileSync(ex.planPath, lines.join("\n"), "utf8");
1300
+ // v0.9.1 (F-008): only a successful plan write mints the live tasks —
1301
+ // pushing before the write left checkpoint entries the plan file does
1302
+ // not contain whenever the write failed.
1303
+ for (const entry of entries) {
1304
+ ex.tasks.push({ id: entry.id, title: entry.title, wave, deps: [], files: [], status: "pending", children: [] });
1305
+ appended.push(entry.id);
1306
+ }
1307
+ } catch {
1308
+ /* best-effort: stranded highs surface via the cap pause */
1309
+ }
1310
+ return appended;
1311
+ }
1312
+
1190
1313
  async function commitReviewOutcome(
1191
1314
  ctx: ExtensionContext,
1192
1315
  ex: ExecState,
1193
1316
  round: InFlightReview,
1194
- outcome: { passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null,
1317
+ outcome: AuditOutcome | null,
1195
1318
  pendingIds: string[],
1196
1319
  ): Promise<void> {
1197
1320
  // The budget is charged only when an outcome commits — never on discard
@@ -1203,106 +1326,169 @@ async function commitReviewOutcome(
1203
1326
  // report omitted the check, spelled the verdict unreadably, or the
1204
1327
  // subagent never ran.
1205
1328
  const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
1329
+ // v0.9: the newest round's reported findings ARE the unresolved set
1330
+ // (stable ids — a problem is resolved only by no longer being reported).
1331
+ // v0.9.1 (F-001): only a REAL parsed report is authoritative. A round that
1332
+ // produced no report at all — spawn failure (outcome === null) or the
1333
+ // two-consecutive-discard synthesis — must PRESERVE the unresolved set:
1334
+ // clearing it let the vacuous completion guard (empty pendingIds) mark a
1335
+ // run done with its high finding silently dropped.
1336
+ const reported = outcome !== null && outcome.findings !== undefined;
1337
+ const findings = reported ? outcome.findings : ex.audit.findings;
1338
+ ex.audit.findings = findings;
1339
+ const highs = findings.filter((f) => f.severity === "high");
1340
+ // Findings are actionable only when this round actually reported them;
1341
+ // see the fix-loop branch below (v0.9.1, F-001).
1342
+ const actionableHighs = reported ? highs : [];
1206
1343
  const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
1207
1344
  const reportText = outcome?.report ?? "(review subagent failed to run)";
1345
+ let reportPath: string | null = null;
1208
1346
  {
1209
1347
  const runDir = runDirOf(ctx);
1210
1348
  if (runDir) {
1211
- writeReviewRoundReport(runDir, {
1349
+ reportPath = writeReviewRoundReport(runDir, {
1212
1350
  budgetRound: round.budgetRound,
1213
1351
  attempt: round.attempt,
1214
1352
  outcome: outcome === null
1215
1353
  ? "spawn-failed"
1216
- : failed.length > 0
1354
+ : failed.length > 0 || actionableHighs.length > 0
1217
1355
  ? "failed"
1218
1356
  : undeterminable.length > 0 ? "undeterminable" : "passed",
1219
1357
  passed,
1220
1358
  failed,
1221
1359
  undeterminable,
1360
+ findings,
1222
1361
  fingerprintCaptured: round.fingerprint,
1223
1362
  coveredTaskIds,
1224
1363
  report: reportText,
1225
1364
  });
1226
1365
  }
1227
1366
  }
1367
+ // Partial progress counts: a check affirmed this round is done even when
1368
+ // a sibling failed, so a later round only re-judges what is still open.
1369
+ for (const id of passed) {
1370
+ const item = ex.items.find((candidate) => candidate.id === id);
1371
+ if (item) item.done = true;
1372
+ }
1373
+
1374
+ // v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
1375
+ // high finding can never complete the run (liveness). Failed checks and
1376
+ // high findings drive ONE union rollback and exactly one executor wake;
1377
+ // high wins over the undeterminable self-schedule (a finding is
1378
+ // actionable independent of verdict evidence). v0.9.1 (F-001): verdicts
1379
+ // are authoritative whenever a report exists, but the FINDINGS-driven
1380
+ // half of the branch needs a findings-bearing report — a no-findings
1381
+ // round (spawn failure, discard synthesis, legacy shape) preserves the
1382
+ // unresolved set and self-schedules instead of rolling back on it.
1383
+ if (failed.length > 0 || actionableHighs.length > 0) {
1384
+ ex.audit.failed = failed;
1385
+ ex.audit.undeterminable = undeterminable;
1386
+ const rolledBack: string[] = [];
1387
+ for (const id of failed) {
1388
+ rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1389
+ }
1390
+ // Invalidation is the VC-fail invariant ONLY: a check re-verifies its
1391
+ // reopened tasks when a FAILED check rolled them back. A pure
1392
+ // finding-driven rollback keeps earlier passes — the findings channel
1393
+ // itself re-examines the repaired work next round (stable ids).
1394
+ if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1395
+ const knownIds = new Set(coveredTaskIds);
1396
+ const mappedHighIds = [...new Set(actionableHighs.flatMap((h) => h.taskIds).filter((id) => knownIds.has(id)))];
1397
+ const highRolledBack = findingsRollbackSet(ex.tasks, mappedHighIds);
1398
+ const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
1399
+ const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
1400
+ const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
1401
+ withExecutionCheckpoint(ctx, (cp) => {
1402
+ // v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
1403
+ // re-stamp the checkpoint's plan identity in the same revision so a
1404
+ // later /resume-plans accepts the amended plan instead of rejecting
1405
+ // it as plan-mismatch (which would cost a full re-approval).
1406
+ const amendedCp = amended.length > 0
1407
+ ? applyExecutionPlanAmended(cp, planIdentityOf(ex.planPath, cp.plan?.version ?? 1), ex.audit.rounds)
1408
+ : cp;
1409
+ return applyExecutionProgress(amendedCp, {
1410
+ tasks: taskProgressMap(ex.tasks),
1411
+ doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1412
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
1413
+ });
1414
+ });
1415
+ if (allRolledBack.length > 0 || amended.length > 0) {
1416
+ // Rolling back (or appending) is itself forward progress for the
1417
+ // watchdog, but NOT for audit.rounds: that counter stays monotonic
1418
+ // so the repair loop is bounded. The repair belongs to the
1419
+ // executor — back to executing.
1420
+ ex.stall.rounds = 0;
1421
+ ex.stall.lastSnapshot = stallSnapshot();
1422
+ setRunStatusForReview(ctx, "executing");
1423
+ }
1424
+ persist(ctx);
1425
+ updateStatusWidget(ctx);
1426
+ // v0.8 wake, generalized (v0.9): per-round one-shot token — exactly one
1427
+ // triggerTurn per committed fix-needing outcome, even when the round
1428
+ // resolves long after the settle that spawned it.
1429
+ if (!round.wakeSent) {
1430
+ round.wakeSent = true;
1431
+ const openTasks = flattenTaskViews(ex.tasks)
1432
+ .filter((task) => !taskIsTerminal(task))
1433
+ .map((task) => task.id);
1434
+ const stranded = allRolledBack.length === 0 && amended.length === 0;
1435
+ const highLines = actionableHighs.map((h) => `- ${h.id}${h.taskIds.length ? ` (${h.taskIds.join(", ")})` : ""}: ${h.note}`).join("\n");
1436
+ const reportRef = reportPath
1437
+ ? `Full round report: ${reportPath}`
1438
+ : `Full round report (run dir unwritable — inline):\n\n---\n${reportText.slice(0, 4000)}`;
1439
+ // v0.9.1 (F-004): a pure VC-fail round keeps the v0.8 lead — never
1440
+ // announce "0 high-severity findings" over an empty block.
1441
+ const findingsLead = actionableHighs.length > 0
1442
+ ? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
1443
+ : `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
1444
+ const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
1445
+ const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
1446
+ messaging().sendMessage(
1447
+ {
1448
+ customType: "pi-plans-audit-failed",
1449
+ content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}`,
1450
+ display: true,
1451
+ },
1452
+ { triggerTurn: true },
1453
+ );
1454
+ }
1455
+ return; // The agent repairs; the next settle re-enters the loop.
1456
+ }
1228
1457
 
1229
- // Fail-closed completion: every pending check affirmatively passed. An
1230
- // all-undeterminable round yields failed === [] — completing here would be
1231
- // fail-open, marking a run done with nothing verified.
1232
- if (passed.length === pendingIds.length) {
1458
+ // Fail-closed completion: every pending check affirmatively passed AND no
1459
+ // high finding remains. The highs guard is explicit (v0.9.1, F-001): a
1460
+ // no-report round skips the fix branch above, so this is the last line
1461
+ // against vacuously completing an empty-pendingIds round with an
1462
+ // unresolved high. An all-undeterminable round yields failed === [] —
1463
+ // completing here would be fail-open, marking a run done with nothing
1464
+ // verified.
1465
+ if (passed.length === pendingIds.length && highs.length === 0) {
1233
1466
  ex.audit.failed = [];
1234
1467
  ex.audit.undeterminable = [];
1235
- for (const id of passed) {
1236
- const item = ex.items.find((candidate) => candidate.id === id);
1237
- if (item) item.done = true;
1238
- }
1239
1468
  withExecutionCheckpoint(ctx, (cp) =>
1240
1469
  applyExecutionProgress(cp, {
1241
1470
  tasks: taskProgressMap(ex.tasks),
1242
1471
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1243
- audit: { rounds: ex.audit.rounds, passed: true },
1472
+ audit: { rounds: ex.audit.rounds, passed: true, findings },
1244
1473
  }),
1245
1474
  );
1246
1475
  await completeExecution(ctx);
1247
1476
  return;
1248
1477
  }
1249
- // Partial progress counts: a check affirmed this round is done even when a
1250
- // sibling failed, so a later round only re-judges what is still open.
1251
- for (const id of passed) {
1252
- const item = ex.items.find((candidate) => candidate.id === id);
1253
- if (item) item.done = true;
1254
- }
1478
+
1479
+ // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1480
+ // wake, no message; the dashboard/overlay carries the round counter.
1255
1481
  ex.audit.failed = failed;
1256
1482
  ex.audit.undeterminable = undeterminable;
1257
- const rolledBack: string[] = [];
1258
- for (const id of failed) {
1259
- rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
1260
- }
1261
- // A rollback reopens work other checks were verifying; those checks must
1262
- // stop claiming the run is satisfied there.
1263
- if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
1264
1483
  withExecutionCheckpoint(ctx, (cp) =>
1265
1484
  applyExecutionProgress(cp, {
1266
1485
  tasks: taskProgressMap(ex.tasks),
1267
1486
  doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
1268
- audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
1487
+ audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
1269
1488
  }),
1270
1489
  );
1271
- if (rolledBack.length > 0) {
1272
- // Rolling back is itself forward progress for the watchdog, but NOT for
1273
- // audit.rounds: that counter stays monotonic so the repair loop is
1274
- // bounded. The repair belongs to the executor — back to executing.
1275
- ex.stall.rounds = 0;
1276
- ex.stall.lastSnapshot = stallSnapshot();
1277
- setRunStatusForReview(ctx, "executing");
1278
- }
1279
1490
  persist(ctx);
1280
1491
  updateStatusWidget(ctx);
1281
- if (failed.length > 0) {
1282
- // v0.8 wake: per-round one-shot token — exactly one triggerTurn per
1283
- // committed failed outcome, even when the round resolves long after the
1284
- // settle that spawned it. The continuation runtime is deliberately
1285
- // untouched: detached rounds outlive their settle.
1286
- if (!round.wakeSent) {
1287
- round.wakeSent = true;
1288
- const openTasks = flattenTaskViews(ex.tasks)
1289
- .filter((task) => !taskIsTerminal(task))
1290
- .map((task) => task.id);
1291
- const stranded = rolledBack.length === 0;
1292
- const content = `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the review automatically.` : ""}`;
1293
- messaging().sendMessage(
1294
- {
1295
- customType: "pi-plans-audit-failed",
1296
- content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}\n\n---\n${reportText.slice(0, 4000)}`,
1297
- display: true,
1298
- },
1299
- { triggerTurn: true },
1300
- );
1301
- }
1302
- return; // The agent repairs; the next settle re-enters the loop.
1303
- }
1304
- // Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
1305
- // wake, no message; the dashboard/overlay carries the round counter.
1306
1492
  await maybeContinueReview(ctx, ex);
1307
1493
  }
1308
1494
 
@@ -1899,7 +2085,8 @@ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
1899
2085
  // This also bounds the zero-input continue loop in agent_before_settle.
1900
2086
  if (ex.stall.paused) return false;
1901
2087
  if (ex.review.inFlight) return false;
1902
- return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
2088
+ return allTasksTerminal(ex.tasks)
2089
+ && (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
1903
2090
  }
1904
2091
 
1905
2092
  /** v0.7.1: record that this settle already ran (or declined) its audit, so a
@@ -2052,10 +2239,12 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
2052
2239
  pausedEx.audit.rounds = 0;
2053
2240
  pausedEx.audit.failed = [];
2054
2241
  pausedEx.audit.undeterminable = [];
2242
+ // v0.9: the fresh budget inherits unresolved findings (stable ids keep
2243
+ // counting) — only the round counter resets.
2055
2244
  withExecutionCheckpoint(ctx, (cp) =>
2056
2245
  applyExecutionProgress(cp, {
2057
2246
  tasks: taskProgressMap(pausedEx.tasks),
2058
- audit: { rounds: 0, lastResult: undefined },
2247
+ audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
2059
2248
  pausedReason: null,
2060
2249
  }),
2061
2250
  );
@@ -2102,6 +2291,12 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2102
2291
  const summary = flat
2103
2292
  .map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
2104
2293
  .join("\n");
2294
+ // v0.9: residual (non-high) findings are summarized, never silent — the
2295
+ // loop converged because nothing high-blocking remained.
2296
+ const residualFindings = execution.audit.findings.filter((f) => f.severity !== "high");
2297
+ const residualNote = residualFindings.length > 0
2298
+ ? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
2299
+ : "";
2105
2300
  const planPath = execution.planPath;
2106
2301
  withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
2107
2302
  execution = null;
@@ -2111,7 +2306,7 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
2111
2306
  messaging().sendMessage(
2112
2307
  {
2113
2308
  customType: "pi-plans-complete",
2114
- content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}`,
2309
+ content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
2115
2310
  display: true,
2116
2311
  },
2117
2312
  { triggerTurn: false },
@@ -2146,13 +2341,17 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
2146
2341
  const rollbackNote = execution.audit.failed.length > 0
2147
2342
  ? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
2148
2343
  : "";
2344
+ const unresolvedHighs = unresolvedHighFindings(execution);
2345
+ const highFindingsNote = unresolvedHighs.length > 0
2346
+ ? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
2347
+ : "";
2149
2348
  return `[PI-PLANS EXECUTION — write access enabled]
2150
2349
  Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
2151
2350
 
2152
2351
  Current wave ${currentWave} open tasks:
2153
2352
  ${waveList}
2154
2353
 
2155
- Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
2354
+ Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}
2156
2355
 
2157
2356
  ${graphLine}
2158
2357
 
@@ -2217,7 +2416,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2217
2416
  planTasks = parsePlanTasks(planText);
2218
2417
  const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
2219
2418
  tasks = buildTaskView(planTasks, snapshotProgress);
2220
- items = parseChecklist(planText);
2419
+ // Fresh text wins, but the satisfied state survives the re-parse: a
2420
+ // mid-loop session restore (v0.9.1, found via F-001's self-schedule
2421
+ // path) must not hand the next round a brief that re-judges checks an
2422
+ // earlier round already passed — that burned budget on every /reload.
2423
+ const snapDone = new Set(snapshot.items.filter((c) => c.done).map((c) => c.id));
2424
+ items = parseChecklist(planText).map((item) => (snapDone.has(item.id) ? { ...item, done: true } : item));
2221
2425
  } catch {
2222
2426
  tasks = snapshot.tasks;
2223
2427
  }
@@ -2238,6 +2442,7 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
2238
2442
  rounds: snapshot.audit?.rounds ?? 0,
2239
2443
  failed: snapshot.audit?.failed ?? [],
2240
2444
  undeterminable: [],
2445
+ findings: toReviewFindings(snapshot.audit?.findings),
2241
2446
  running: false,
2242
2447
  },
2243
2448
  review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },