pi-plans 0.7.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +92 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +14 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +306 -63
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +91 -13
- package/src/exec.ts +835 -142
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +40 -8
- package/src/resume-command.ts +19 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +62 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +93 -6
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +299 -16
- package/tests/dashboard.test.ts +202 -2
- package/tests/exec-review-loop.test.ts +724 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/refine-ui.test.ts +25 -2
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +105 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
- package/tools/refine.ts +22 -3
package/src/exec.ts
CHANGED
|
@@ -9,15 +9,16 @@
|
|
|
9
9
|
* live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
|
|
10
10
|
* tree), a stall watchdog pauses the run when consecutive rounds produce no
|
|
11
11
|
* task-state change, and when every task reaches a terminal state an
|
|
12
|
-
* independent
|
|
13
|
-
* failed checks roll their covered tasks
|
|
14
|
-
* channel), and
|
|
15
|
-
*
|
|
12
|
+
* independent execution reviewer verifies the plan's verification checks in
|
|
13
|
+
* a detached, overlay-visible loop — failed checks roll their covered tasks
|
|
14
|
+
* back to pending (audit-flow-only channel), and five committed rounds pause
|
|
15
|
+
* the run for the user in every mode (fail-closed, never a silent stop).
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import * as fs from "node:fs";
|
|
19
19
|
import * as path from "node:path";
|
|
20
|
-
import { randomUUID } from "node:crypto";
|
|
20
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
21
|
+
import { execSync } from "node:child_process";
|
|
21
22
|
import type {
|
|
22
23
|
CompactionResult,
|
|
23
24
|
ExtensionAPI,
|
|
@@ -44,7 +45,7 @@ import {
|
|
|
44
45
|
type VccCompactionBuildResult,
|
|
45
46
|
type VccCompactionStats,
|
|
46
47
|
} from "./compaction.ts";
|
|
47
|
-
import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
48
|
+
import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, runDirPath, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
48
49
|
import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
|
|
49
50
|
import { bindRun, resolveActiveRun } from "./run-context.ts";
|
|
50
51
|
import type { SubagentProgressEvent } from "./subagent.ts";
|
|
@@ -56,6 +57,7 @@ import {
|
|
|
56
57
|
applyExecutionProgress,
|
|
57
58
|
applyExecutionStopped,
|
|
58
59
|
createCheckpoint,
|
|
60
|
+
applyExecutionPlanAmended,
|
|
59
61
|
loadCheckpoint,
|
|
60
62
|
mutateCheckpoint,
|
|
61
63
|
planIdentityOf,
|
|
@@ -71,6 +73,7 @@ import { resolveGraphMode } from "./code-graph/mode.ts";
|
|
|
71
73
|
import {
|
|
72
74
|
parseChecklist,
|
|
73
75
|
parsePlanTasks,
|
|
76
|
+
CHECKLIST_HEADERS,
|
|
74
77
|
flattenTasks,
|
|
75
78
|
type CheckItem,
|
|
76
79
|
type PlanTasks,
|
|
@@ -81,7 +84,10 @@ import {
|
|
|
81
84
|
auditableChecks,
|
|
82
85
|
buildTaskView,
|
|
83
86
|
currentTask,
|
|
87
|
+
findingsRollbackSet,
|
|
84
88
|
flattenTaskViews,
|
|
89
|
+
invalidateChecksForRolledBackTasks,
|
|
90
|
+
maxWave,
|
|
85
91
|
taskIsTerminal,
|
|
86
92
|
taskProgress,
|
|
87
93
|
taskProgressMap,
|
|
@@ -97,8 +103,15 @@ import {
|
|
|
97
103
|
renderDashboardLines,
|
|
98
104
|
renderDashboardTreeLines,
|
|
99
105
|
} from "./dashboard.ts";
|
|
100
|
-
import {
|
|
106
|
+
import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
|
|
107
|
+
import type { ReviewFindingRecord } from "./workflow-state.ts";
|
|
108
|
+
import { staleReloadHint as probeStaleReload } from "./staleness.ts";
|
|
101
109
|
import { messaging } from "./messaging.ts";
|
|
110
|
+
import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
|
|
111
|
+
import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
|
|
112
|
+
import { resolveReviewerSpawn } from "./thinking-levels.ts";
|
|
113
|
+
import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
|
|
114
|
+
import { matchesTerminalKey } from "./terminal-keys.ts";
|
|
102
115
|
|
|
103
116
|
export interface ExecState {
|
|
104
117
|
planPath: string;
|
|
@@ -117,8 +130,14 @@ export interface ExecState {
|
|
|
117
130
|
/** Stall watchdog (v0.6.1): consecutive settled rounds without a task
|
|
118
131
|
* status change; auto-pause at the cap. */
|
|
119
132
|
stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
|
|
120
|
-
/** Completion-audit bookkeeping.
|
|
121
|
-
|
|
133
|
+
/** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
|
|
134
|
+
* only when a round outcome commits (never on discard/cancel) — and is the
|
|
135
|
+
* only piece persisted. */
|
|
136
|
+
audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean; findings: ReviewFinding[] };
|
|
137
|
+
/** Execution-review loop (v0.8), memory-only: the attempt index names the
|
|
138
|
+
* per-round report files; consecutiveDiscards bounds the fingerprint
|
|
139
|
+
* re-run loop; inFlight owns the round's abort lifecycle. */
|
|
140
|
+
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
|
|
122
141
|
/** Per-settle audit latch (v0.7.1): a settled round fires the completion
|
|
123
142
|
* audit at most once, so the turn_end / agent_before_settle / resume entry
|
|
124
143
|
* points cannot double-consume a round when several land in one settle.
|
|
@@ -126,6 +145,18 @@ export interface ExecState {
|
|
|
126
145
|
auditLatch?: { auditedThisSettle: boolean; activity: number };
|
|
127
146
|
}
|
|
128
147
|
|
|
148
|
+
/** One in-flight review round: owns its abort lifecycle, its fingerprint of
|
|
149
|
+
* the audited subject, and the per-round one-shot wake token (v0.8). */
|
|
150
|
+
interface InFlightReview {
|
|
151
|
+
controller: AbortController;
|
|
152
|
+
/** Budget round this attempt belongs to (audit.rounds + 1 at spawn). */
|
|
153
|
+
budgetRound: number;
|
|
154
|
+
/** Monotonic attempt ordinal; names the round report file. */
|
|
155
|
+
attempt: number;
|
|
156
|
+
fingerprint: string;
|
|
157
|
+
wakeSent: boolean;
|
|
158
|
+
}
|
|
159
|
+
|
|
129
160
|
/**
|
|
130
161
|
* D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
|
|
131
162
|
* and status bar right after a `set-language` change.
|
|
@@ -204,6 +235,10 @@ export interface CheckpointExecutionLoad {
|
|
|
204
235
|
/** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
|
|
205
236
|
* detected — resume requires a fresh handoff approval. */
|
|
206
237
|
legacyDelegate?: boolean;
|
|
238
|
+
/** v0.9.1 (F-005): unresolved findings from the newest committed round,
|
|
239
|
+
* so the /resume-plans brief can surface outstanding highs before the
|
|
240
|
+
* per-turn injection ever runs. */
|
|
241
|
+
findings?: ReviewFinding[];
|
|
207
242
|
error?: string;
|
|
208
243
|
}
|
|
209
244
|
|
|
@@ -253,9 +288,12 @@ export function loadExecutionFromCheckpoint(
|
|
|
253
288
|
const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
|
|
254
289
|
const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
|
|
255
290
|
const tasks = buildTaskView(planTasks, progress);
|
|
256
|
-
//
|
|
257
|
-
//
|
|
258
|
-
|
|
291
|
+
// v0.8: a review-cap pause SURVIVES the restore — the budget must stay
|
|
292
|
+
// bounded across restarts; only /plans-execute grants a fresh one. The
|
|
293
|
+
// legacy v0.7 prefix stays dual-matched for one release.
|
|
294
|
+
const wasReviewCapPause = isReviewCapPause(cp.execution.pausedReason);
|
|
295
|
+
// Any live round from the replaced session graph dies here (CF2-002).
|
|
296
|
+
abortInFlightReview();
|
|
259
297
|
if (!reverifyAll) {
|
|
260
298
|
for (const id of cp.execution.doneVcIds) {
|
|
261
299
|
const item = items.find((candidate) => candidate.id === id);
|
|
@@ -271,22 +309,25 @@ export function loadExecutionFromCheckpoint(
|
|
|
271
309
|
startedAt: utcNow(),
|
|
272
310
|
usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
|
|
273
311
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
274
|
-
// D-020: a paused legacy (or stopped) execution rebuilds unpaused —
|
|
275
|
-
//
|
|
276
|
-
//
|
|
277
|
-
//
|
|
278
|
-
//
|
|
279
|
-
|
|
280
|
-
rounds: 0,
|
|
312
|
+
// D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
|
|
313
|
+
// resume itself is the user's intent; the reason is surfaced in the
|
|
314
|
+
// resume brief instead. EXCEPT a review-cap pause (v0.8): it must
|
|
315
|
+
// survive restores paused and at its committed round count, or the
|
|
316
|
+
// 5-round budget would never bound anything across restarts.
|
|
317
|
+
stall: {
|
|
318
|
+
rounds: cp.execution.stallRounds ?? 0,
|
|
281
319
|
lastSnapshot: null,
|
|
282
|
-
paused:
|
|
283
|
-
pausedReason: undefined,
|
|
320
|
+
paused: wasReviewCapPause,
|
|
321
|
+
pausedReason: wasReviewCapPause ? (cp.execution.pausedReason ?? undefined) : undefined,
|
|
284
322
|
},
|
|
285
323
|
audit: {
|
|
286
|
-
rounds:
|
|
324
|
+
rounds: cp.execution.audit?.rounds ?? 0,
|
|
287
325
|
failed: [],
|
|
326
|
+
undeterminable: cp.execution.audit?.undeterminable ?? [],
|
|
327
|
+
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
288
328
|
running: false,
|
|
289
329
|
},
|
|
330
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
290
331
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
291
332
|
};
|
|
292
333
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
@@ -295,11 +336,6 @@ export function loadExecutionFromCheckpoint(
|
|
|
295
336
|
resetContinuationRuntime(ctx);
|
|
296
337
|
pendingExecutionFlush = false;
|
|
297
338
|
resetExecutionCompactionState(ctx);
|
|
298
|
-
if (wasAuditCapPause) {
|
|
299
|
-
withExecutionCheckpoint(ctx, (cp2) =>
|
|
300
|
-
applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
|
|
301
|
-
);
|
|
302
|
-
}
|
|
303
339
|
// D-020: an orphaned v0.6.0 delegated executor never survives a restart.
|
|
304
340
|
// Its checkpoint delegate marker REFUSES the direct load — the run must
|
|
305
341
|
// re-enter through the execution handoff so the C-006 approval gate
|
|
@@ -337,6 +373,7 @@ export function loadExecutionFromCheckpoint(
|
|
|
337
373
|
pausedReason: cp.execution.pausedReason,
|
|
338
374
|
legacyPlan: planTasks.legacy,
|
|
339
375
|
legacyDelegate,
|
|
376
|
+
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
340
377
|
};
|
|
341
378
|
}
|
|
342
379
|
|
|
@@ -475,6 +512,9 @@ function updatePanelWidget(ctx: ExtensionContext): void {
|
|
|
475
512
|
pausedReason: current.stall.pausedReason,
|
|
476
513
|
auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
|
|
477
514
|
auditFailed: current.audit.failed,
|
|
515
|
+
auditUndeterminable: current.audit.undeterminable,
|
|
516
|
+
findings: current.audit.findings,
|
|
517
|
+
reviewRunning: current.audit.running === true || current.review.inFlight !== null,
|
|
478
518
|
startedAt: current.startedAt,
|
|
479
519
|
usage: current.usage,
|
|
480
520
|
});
|
|
@@ -497,6 +537,9 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
497
537
|
pausedReason: execution.stall.pausedReason,
|
|
498
538
|
auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
|
|
499
539
|
auditFailed: execution.audit.failed,
|
|
540
|
+
auditUndeterminable: execution.audit.undeterminable,
|
|
541
|
+
findings: execution.audit.findings,
|
|
542
|
+
reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
|
|
500
543
|
});
|
|
501
544
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
|
|
502
545
|
return;
|
|
@@ -524,6 +567,10 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
524
567
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
|
|
525
568
|
return;
|
|
526
569
|
}
|
|
570
|
+
if (status === "verifying") {
|
|
571
|
+
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `🔎 plans: ${active.run_id} (verifying)`));
|
|
572
|
+
return;
|
|
573
|
+
}
|
|
527
574
|
if (status === "planning") {
|
|
528
575
|
const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
|
|
529
576
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
|
|
@@ -564,7 +611,7 @@ function persist(ctx: ExtensionContext): void {
|
|
|
564
611
|
startedAt: execution.startedAt,
|
|
565
612
|
usage: execution.usage,
|
|
566
613
|
stall: execution.stall,
|
|
567
|
-
audit: { rounds: execution.audit.rounds, failed: execution.audit.failed },
|
|
614
|
+
audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
|
|
568
615
|
});
|
|
569
616
|
}
|
|
570
617
|
|
|
@@ -594,6 +641,8 @@ export async function startExecution(
|
|
|
594
641
|
ctx: ExtensionContext,
|
|
595
642
|
input: StartExecutionInput,
|
|
596
643
|
): Promise<void> {
|
|
644
|
+
// A fresh handoff replaces any live run — abort its in-flight review round first.
|
|
645
|
+
abortInFlightReview();
|
|
597
646
|
const tasks = buildTaskView(input.planTasks);
|
|
598
647
|
execution = {
|
|
599
648
|
planPath: input.planPath,
|
|
@@ -605,7 +654,8 @@ export async function startExecution(
|
|
|
605
654
|
usage: { inToks: 0, outToks: 0 },
|
|
606
655
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
607
656
|
stall: { rounds: 0, lastSnapshot: null, paused: false },
|
|
608
|
-
audit: { rounds: 0, failed: [], running: false },
|
|
657
|
+
audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
|
|
658
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
609
659
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
610
660
|
};
|
|
611
661
|
// Seed the watchdog baseline only after `execution` points at the new state
|
|
@@ -670,9 +720,14 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
|
|
|
670
720
|
applyExecutionProgress(cp, {
|
|
671
721
|
tasks: taskProgressMap(execution!.tasks),
|
|
672
722
|
doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
|
|
723
|
+
stallRounds: execution!.stall.rounds,
|
|
673
724
|
audit: {
|
|
674
725
|
rounds: execution!.audit.rounds,
|
|
675
726
|
lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
|
|
727
|
+
undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
|
|
728
|
+
// v0.9: audit writes are replace-semantics — every writer must
|
|
729
|
+
// carry findings or a task update would silently wipe them.
|
|
730
|
+
findings: execution!.audit.findings,
|
|
676
731
|
},
|
|
677
732
|
}),
|
|
678
733
|
);
|
|
@@ -699,10 +754,10 @@ export function recordExecutionTurn(
|
|
|
699
754
|
}
|
|
700
755
|
|
|
701
756
|
/** Test hook: replace the audit subagent with a deterministic function. */
|
|
702
|
-
let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
|
|
757
|
+
let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
|
|
703
758
|
|
|
704
759
|
export function __setAuditRunnerForTests(
|
|
705
|
-
runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
|
|
760
|
+
runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
|
|
706
761
|
): void {
|
|
707
762
|
auditRunnerForTests = runner;
|
|
708
763
|
}
|
|
@@ -746,15 +801,17 @@ export function registerExecutionTurnHandlers(
|
|
|
746
801
|
maybeContinuationFollowUp(ctx);
|
|
747
802
|
return;
|
|
748
803
|
}
|
|
749
|
-
// Fully settled and still owed
|
|
750
|
-
//
|
|
751
|
-
//
|
|
752
|
-
//
|
|
804
|
+
// Fully settled and still owed a review: launch the round under the
|
|
805
|
+
// mode rule — tui/rpc detach (the settle returns NOW and the overlay
|
|
806
|
+
// carries progress); print/json await inline so runtime teardown cannot
|
|
807
|
+
// kill the child. The round's own outcome routing drives the rest.
|
|
753
808
|
latchAuditThisSettle();
|
|
754
|
-
|
|
809
|
+
const chain = launchReviewRound(ctx);
|
|
810
|
+
if (chain) await chain;
|
|
755
811
|
});
|
|
756
812
|
ext.on("session_shutdown", async (_event, ctx) => {
|
|
757
813
|
drainExecutionFlush(ctx);
|
|
814
|
+
abortInFlightReview();
|
|
758
815
|
execution = null;
|
|
759
816
|
executionRunId = null;
|
|
760
817
|
continuationRuntime = null;
|
|
@@ -795,20 +852,242 @@ export function registerExecutionTurnHandlers(
|
|
|
795
852
|
if (usage) recordExecutionTurn(ctx, usage);
|
|
796
853
|
if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
|
|
797
854
|
latchAuditThisSettle();
|
|
798
|
-
|
|
855
|
+
const chain = launchReviewRound(ctx);
|
|
856
|
+
if (chain) await chain;
|
|
799
857
|
}
|
|
800
858
|
await onTurnEnd?.(ctx);
|
|
801
859
|
});
|
|
802
860
|
}
|
|
803
861
|
|
|
804
|
-
/**
|
|
805
|
-
* complete the run, keep iterating (rollback), pause at the round cap
|
|
806
|
-
* (interactive), or stop at the cap (auto-approve/headless — D-022).
|
|
862
|
+
/** ==== Execution-review loop (v0.8) ====
|
|
807
863
|
*
|
|
808
|
-
*
|
|
809
|
-
*
|
|
810
|
-
*
|
|
811
|
-
|
|
864
|
+
* When every task is terminal and checks are still owed, the run enters the
|
|
865
|
+
* `verifying` status and a DETACHED read-only reviewer round runs in the
|
|
866
|
+
* background (tui/rpc): the settle handler returns immediately and the
|
|
867
|
+
* executor is truly idle while the overlay shows live progress. In print/json
|
|
868
|
+
* modes the settle handler keeps AWAITING the round inline — runtime
|
|
869
|
+
* teardown at settle would otherwise kill a detached child and swallow the
|
|
870
|
+
* pause signal.
|
|
871
|
+
*
|
|
872
|
+
* Budget: `audit.rounds` counts COMMITTED rounds only (pass, fail, or
|
|
873
|
+
* undeterminable); discards and cancellations burn nothing. Undeterminable
|
|
874
|
+
* rounds self-schedule the retry inside the loop (no wake). Two consecutive
|
|
875
|
+
* fingerprint discards commit as an undeterminable round so the loop stays
|
|
876
|
+
* bounded. Exhaustion pauses in EVERY mode (fail-closed) with an in-band
|
|
877
|
+
* `pi-plans-review-paused` message; the ONLY fresh-budget surface is
|
|
878
|
+
* /plans-execute — ordinary input and session restores never refill.
|
|
879
|
+
*
|
|
880
|
+
* Lifecycle: each round owns a session-scoped AbortController (never
|
|
881
|
+
* ctx.signal, which is turn-scoped), aborted from session_shutdown,
|
|
882
|
+
* stopExecution, startExecution, and restoreFromSession. Outcomes are
|
|
883
|
+
* guarded by execution identity (`execution !== owner` → silent discard).
|
|
884
|
+
*
|
|
885
|
+
* Completion stays fail-closed AND never fail-open: a run completes only
|
|
886
|
+
* when every pending check was affirmatively passed. */
|
|
887
|
+
const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
|
|
888
|
+
/** v0.7 protocol value — dual-matched for one release so checkpoints written
|
|
889
|
+
* by older builds keep their cap pause recognized on restore. */
|
|
890
|
+
const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
891
|
+
|
|
892
|
+
function isReviewCapPause(reason: string | undefined | null): boolean {
|
|
893
|
+
if (!reason) return false;
|
|
894
|
+
return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
/** The currently-running (or self-scheduling) review chain; the sanctioned
|
|
898
|
+
* test seam awaits this. */
|
|
899
|
+
let activeReviewChain: Promise<void> | null = null;
|
|
900
|
+
|
|
901
|
+
function abortInFlightReview(): void {
|
|
902
|
+
const inFlight = execution?.review.inFlight;
|
|
903
|
+
if (inFlight) {
|
|
904
|
+
try {
|
|
905
|
+
inFlight.controller.abort();
|
|
906
|
+
} catch {
|
|
907
|
+
/* already aborted */
|
|
908
|
+
}
|
|
909
|
+
if (execution) execution.review.inFlight = null;
|
|
910
|
+
}
|
|
911
|
+
activeReviewChain = null;
|
|
912
|
+
}
|
|
913
|
+
|
|
914
|
+
/** v0.9: unresolved high-severity findings from the newest committed round.
|
|
915
|
+
* Presence in the newest round's report IS the unresolved set (stable ids:
|
|
916
|
+
* a problem is resolved only by no longer being reported). */
|
|
917
|
+
function unresolvedHighFindings(ex: ExecState): ReviewFinding[] {
|
|
918
|
+
return ex.audit.findings.filter((f) => f.severity === "high");
|
|
919
|
+
}
|
|
920
|
+
|
|
921
|
+
/** v0.9: checkpoint records -> runtime findings. Invalid severities degrade to
|
|
922
|
+
* "malformed" (recorded, non-blocking) — the same never-fail posture as the
|
|
923
|
+
* report parser. */
|
|
924
|
+
function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
|
|
925
|
+
if (!records) return [];
|
|
926
|
+
return records.map((r) => {
|
|
927
|
+
const severity = (r.severity === "high" || r.severity === "medium" || r.severity === "low") ? r.severity : "malformed";
|
|
928
|
+
return { id: r.id, severity, taskIds: Array.isArray(r.taskIds) ? r.taskIds : [], proposedTask: r.proposedTask, note: r.note ?? "", evidence: r.evidence ?? "", raw: r.raw ?? "" };
|
|
929
|
+
});
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
/** v0.9: findings widen the owed predicate — an unresolved high finding keeps
|
|
933
|
+
* the review owed even when every check is done (the stranded-high path),
|
|
934
|
+
* mirroring how a failed check keeps it owed today. */
|
|
935
|
+
function reviewOwed(ex: ExecState): boolean {
|
|
936
|
+
return allTasksTerminal(ex.tasks)
|
|
937
|
+
&& (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
function runDirOf(ctx: ExtensionContext): string | null {
|
|
941
|
+
const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
|
|
942
|
+
return runId ? runDirPath(ctx.cwd, runId) : null;
|
|
943
|
+
}
|
|
944
|
+
|
|
945
|
+
function setRunStatusForReview(ctx: ExtensionContext, status: "verifying" | "executing"): void {
|
|
946
|
+
const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
|
|
947
|
+
if (!runId) return;
|
|
948
|
+
try {
|
|
949
|
+
const current = getRun(ctx.cwd, runId)?.status;
|
|
950
|
+
if (current !== status && current !== "done" && current !== "abandoned") {
|
|
951
|
+
setRunStatus(ctx.cwd, runId, status);
|
|
952
|
+
}
|
|
953
|
+
} catch {
|
|
954
|
+
/* best-effort */
|
|
955
|
+
}
|
|
956
|
+
}
|
|
957
|
+
|
|
958
|
+
/** Round fingerprint (Q-fingerprint-scope): plan digest + git HEAD +
|
|
959
|
+
* covered-file mtimes. A change between round start and resolve means the
|
|
960
|
+
* reviewer judged a subject that no longer exists — discard and re-run. */
|
|
961
|
+
function captureReviewFingerprint(ctx: ExtensionContext, ex: ExecState): string {
|
|
962
|
+
const parts: string[] = [];
|
|
963
|
+
try {
|
|
964
|
+
parts.push(createHash("sha256").update(fs.readFileSync(ex.planPath, "utf8")).digest("hex"));
|
|
965
|
+
} catch {
|
|
966
|
+
parts.push("plan-unreadable");
|
|
967
|
+
}
|
|
968
|
+
try {
|
|
969
|
+
parts.push(execSync("git rev-parse HEAD", { cwd: ctx.cwd, stdio: ["ignore", "pipe", "pipe"] }).toString().trim());
|
|
970
|
+
} catch {
|
|
971
|
+
parts.push("no-head");
|
|
972
|
+
}
|
|
973
|
+
const covered = new Set<string>();
|
|
974
|
+
for (const task of flattenTaskViews(ex.tasks)) {
|
|
975
|
+
for (const file of task.files ?? []) covered.add(file);
|
|
976
|
+
}
|
|
977
|
+
const mtimes: string[] = [];
|
|
978
|
+
for (const file of [...covered].sort()) {
|
|
979
|
+
try {
|
|
980
|
+
mtimes.push(`${file}:${fs.statSync(path.resolve(ctx.cwd, file)).mtimeMs}`);
|
|
981
|
+
} catch {
|
|
982
|
+
mtimes.push(`${file}:missing`);
|
|
983
|
+
}
|
|
984
|
+
}
|
|
985
|
+
parts.push(mtimes.join("|"));
|
|
986
|
+
return createHash("sha256").update(parts.join("\u0000")).digest("hex");
|
|
987
|
+
}
|
|
988
|
+
|
|
989
|
+
/** Round timeout: a committed round is minutes, never the 60-min subagent
|
|
990
|
+
* default — a hung child must surface as a spawn-failure round, not park the
|
|
991
|
+
* run in verifying for an hour (CF2-010). */
|
|
992
|
+
const REVIEW_ROUND_TIMEOUT_MS = 20 * 60 * 1000;
|
|
993
|
+
|
|
994
|
+
/** Reviewer-role pinning (Q-role-fallback): a CONFIRMED delegated role pins
|
|
995
|
+
* the spawn's model+thinking and labels the overlay with the role; an
|
|
996
|
+
* unconfirmed or current-session role inherits the session default with the
|
|
997
|
+
* "session default" label — a detached round NEVER opens the interactive
|
|
998
|
+
* first-use panel. */
|
|
999
|
+
function reviewSpawnProfile(): { model?: string; thinkingLevel?: string; label: string } {
|
|
1000
|
+
try {
|
|
1001
|
+
const reviewer = loadGlobalConfig().config.reviewer;
|
|
1002
|
+
if (reviewer.mode !== "current-session" && reviewerReady(reviewer)) {
|
|
1003
|
+
const spawn = resolveReviewerSpawn(reviewer);
|
|
1004
|
+
if (spawn.modelSelector) {
|
|
1005
|
+
return { model: spawn.modelSelector, thinkingLevel: spawn.thinkingLevel ?? undefined, label: spawn.label };
|
|
1006
|
+
}
|
|
1007
|
+
}
|
|
1008
|
+
} catch {
|
|
1009
|
+
/* fall through to the session default */
|
|
1010
|
+
}
|
|
1011
|
+
return { label: "session default" };
|
|
1012
|
+
}
|
|
1013
|
+
|
|
1014
|
+
/** Engine-held lane state for the in-flight round: the reopen path builds a
|
|
1015
|
+
* fresh one-shot controller seeded from THIS object, so the accumulated
|
|
1016
|
+
* transcript survives ESC + reopen (CF2-001 / Q-reopen-seed). */
|
|
1017
|
+
let reviewLane: RefineLaneState | null = null;
|
|
1018
|
+
let reviewOverlay: RefineOverlayController | null = null;
|
|
1019
|
+
let reviewModelLabel: string | undefined;
|
|
1020
|
+
|
|
1021
|
+
function freshReviewLane(attempt: number): RefineLaneState {
|
|
1022
|
+
return {
|
|
1023
|
+
id: `review-round-${attempt}`,
|
|
1024
|
+
label: `Execution review round ${attempt}`,
|
|
1025
|
+
status: "queued",
|
|
1026
|
+
phase: "queued",
|
|
1027
|
+
detail: "",
|
|
1028
|
+
transcript: [],
|
|
1029
|
+
currentTurnIndex: 0,
|
|
1030
|
+
scrollOffset: 0,
|
|
1031
|
+
followTranscript: true,
|
|
1032
|
+
viewportHeight: 1,
|
|
1033
|
+
};
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
/** Fresh controller per round (the controller is one-shot: closed latch,
|
|
1037
|
+
* overlayPromise bail, terminal-lane early return — reuse drops progress).
|
|
1038
|
+
* A UI failure must NEVER kill the round itself — best-effort only. The
|
|
1039
|
+
* overlay forwards unhandled keys so Ctrl+Shift+T keeps working while the
|
|
1040
|
+
* review overlay holds focus (pi-tui has no key bubbling). */
|
|
1041
|
+
function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
|
|
1042
|
+
if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
|
|
1043
|
+
try {
|
|
1044
|
+
const controller = new RefineOverlayController(
|
|
1045
|
+
"auditor",
|
|
1046
|
+
[{ id: lane.id, label: lane.label }],
|
|
1047
|
+
() => {},
|
|
1048
|
+
lang ?? "en",
|
|
1049
|
+
(data) => {
|
|
1050
|
+
if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
|
|
1051
|
+
},
|
|
1052
|
+
);
|
|
1053
|
+
controller.seedLane(lane);
|
|
1054
|
+
controller.open(refineOverlayContext(ctx), modelLabel);
|
|
1055
|
+
return controller;
|
|
1056
|
+
} catch {
|
|
1057
|
+
return null;
|
|
1058
|
+
}
|
|
1059
|
+
}
|
|
1060
|
+
|
|
1061
|
+
/** The reopen surface (Task-3.4): rebuilds the overlay from engine-held lane
|
|
1062
|
+
* state; inert when no round is in flight. */
|
|
1063
|
+
export function reopenReviewOverlay(ctx: ExtensionContext): void {
|
|
1064
|
+
if (!execution?.review.inFlight || !reviewLane) return;
|
|
1065
|
+
// Never stack a second overlay on a live one (Ctrl+Shift+R while open).
|
|
1066
|
+
if (reviewOverlay && !reviewOverlay.isClosed()) return;
|
|
1067
|
+
const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
|
|
1068
|
+
if (controller) reviewOverlay = controller;
|
|
1069
|
+
}
|
|
1070
|
+
|
|
1071
|
+
function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
1072
|
+
const highs = unresolvedHighFindings(ex);
|
|
1073
|
+
const detail =
|
|
1074
|
+
ex.audit.failed.length > 0
|
|
1075
|
+
? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
|
|
1076
|
+
: highs.length > 0
|
|
1077
|
+
? `high findings: ${highs.map((f) => f.id).join(", ")}`
|
|
1078
|
+
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1079
|
+
const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
|
|
1080
|
+
pauseForStall(ctx, reason);
|
|
1081
|
+
// In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
|
|
1082
|
+
// pauseForStall's ui.notify is optional and absent headless, so the pause
|
|
1083
|
+
// must also land in the session stream every mode can read.
|
|
1084
|
+
messaging().sendMessage(
|
|
1085
|
+
{ customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
|
|
1086
|
+
{ triggerTurn: false },
|
|
1087
|
+
);
|
|
1088
|
+
}
|
|
1089
|
+
|
|
1090
|
+
async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
812
1091
|
if (!execution) return;
|
|
813
1092
|
const ex = execution;
|
|
814
1093
|
// Skipped-pass checks resolve without a subagent round.
|
|
@@ -819,99 +1098,451 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
|
|
|
819
1098
|
// Only auditable checks (with task coverage) gate completion; checks that
|
|
820
1099
|
// cover no task can never be verified and never block or complete.
|
|
821
1100
|
const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
|
|
822
|
-
|
|
1101
|
+
// v0.9: unresolved high findings block the fast completion path — they
|
|
1102
|
+
// keep the review owed instead (liveness: a high can never be completed
|
|
1103
|
+
// around, only fixed or paused at the cap).
|
|
1104
|
+
if (pendingChecks.length === 0 && unresolvedHighFindings(ex).length === 0) {
|
|
823
1105
|
await completeExecution(ctx);
|
|
824
1106
|
return;
|
|
825
1107
|
}
|
|
826
|
-
if (ex.
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
if (isInteractiveSession(ctx)) {
|
|
830
|
-
pauseForStall(
|
|
831
|
-
ctx,
|
|
832
|
-
`${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
|
|
833
|
-
);
|
|
834
|
-
return;
|
|
835
|
-
}
|
|
836
|
-
await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
|
|
1108
|
+
if (ex.stall.paused || ex.review.inFlight) return;
|
|
1109
|
+
if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
|
|
1110
|
+
pauseReviewCap(ctx, ex);
|
|
837
1111
|
return;
|
|
838
1112
|
}
|
|
839
|
-
|
|
840
|
-
|
|
1113
|
+
// Phase transition: the executor is done with its tasks; the review loop
|
|
1114
|
+
// owns the run until it converges (or pauses at the cap).
|
|
1115
|
+
setRunStatusForReview(ctx, "verifying");
|
|
1116
|
+
const round: InFlightReview = {
|
|
1117
|
+
controller: new AbortController(),
|
|
1118
|
+
budgetRound: ex.audit.rounds + 1,
|
|
1119
|
+
attempt: ex.review.attempts + 1,
|
|
1120
|
+
fingerprint: captureReviewFingerprint(ctx, ex),
|
|
1121
|
+
wakeSent: false,
|
|
1122
|
+
};
|
|
1123
|
+
ex.review.attempts = round.attempt;
|
|
1124
|
+
ex.review.inFlight = round;
|
|
1125
|
+
ex.audit.running = true; // dashboard mirror (the v0.8 model lands with the overlay task)
|
|
1126
|
+
const spawn = reviewSpawnProfile();
|
|
1127
|
+
reviewModelLabel = spawn.label;
|
|
1128
|
+
const lane = freshReviewLane(round.attempt);
|
|
1129
|
+
reviewLane = lane;
|
|
1130
|
+
reviewOverlay = openReviewOverlay(ctx, lane, ex.uiLanguage, spawn.label);
|
|
841
1131
|
updateStatusWidget(ctx);
|
|
842
|
-
let
|
|
1132
|
+
let result: AuditRoundResult;
|
|
843
1133
|
try {
|
|
844
|
-
|
|
845
|
-
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: ex.audit.
|
|
1134
|
+
result = auditRunnerForTests
|
|
1135
|
+
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt, priorFindings: ex.audit.findings })
|
|
846
1136
|
: await runCompletionAudit(ctx, {
|
|
847
1137
|
planPath: ex.planPath,
|
|
848
1138
|
checklist: ex.items,
|
|
849
1139
|
tasks: ex.tasks,
|
|
850
|
-
round:
|
|
851
|
-
|
|
1140
|
+
round: round.attempt,
|
|
1141
|
+
priorFindings: ex.audit.findings,
|
|
1142
|
+
model: spawn.model,
|
|
1143
|
+
thinkingLevel: spawn.thinkingLevel,
|
|
1144
|
+
timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
|
|
1145
|
+
signal: round.controller.signal,
|
|
1146
|
+
onProgress: (event) => {
|
|
1147
|
+
// The engine owns the lane state; the live controller only repaints.
|
|
1148
|
+
applyRefineProgress(lane, event);
|
|
1149
|
+
reviewOverlay?.rerender();
|
|
1150
|
+
},
|
|
1151
|
+
});
|
|
1152
|
+
} catch (error) {
|
|
1153
|
+
if (ex.review.inFlight === round) {
|
|
1154
|
+
ex.review.inFlight = null;
|
|
1155
|
+
ex.audit.running = false;
|
|
1156
|
+
}
|
|
1157
|
+
void reviewOverlay?.close();
|
|
1158
|
+
reviewOverlay = null;
|
|
1159
|
+
messaging().sendMessage(
|
|
1160
|
+
{ customType: "pi-plans-review-error", content: `pi-plans: execution review round threw: ${String(error)}`, display: true },
|
|
1161
|
+
{ triggerTurn: false },
|
|
1162
|
+
);
|
|
1163
|
+
updateStatusWidget(ctx);
|
|
1164
|
+
return;
|
|
1165
|
+
}
|
|
1166
|
+
// Overlay terminal state + close (the controller is one-shot; the engine-held
|
|
1167
|
+
// lane keeps the transcript for a later reopen within this round).
|
|
1168
|
+
const cancelledResult = result !== null && typeof result === "object" && "cancelled" in result;
|
|
1169
|
+
try {
|
|
1170
|
+
applyRefineResult(lane, {
|
|
1171
|
+
ok: !cancelledResult,
|
|
1172
|
+
output: result && "report" in result ? result.report : "",
|
|
1173
|
+
stderr: "",
|
|
1174
|
+
turns: 0,
|
|
1175
|
+
...(cancelledResult ? { cancelled: true as const } : {}),
|
|
1176
|
+
});
|
|
1177
|
+
} catch {
|
|
1178
|
+
/* cosmetic only */
|
|
1179
|
+
}
|
|
1180
|
+
void reviewOverlay?.close();
|
|
1181
|
+
reviewOverlay = null;
|
|
1182
|
+
await handleReviewOutcome(ctx, ex, round, result, pendingChecks.map((item) => item.id));
|
|
1183
|
+
}
|
|
1184
|
+
|
|
1185
|
+
async function handleReviewOutcome(
|
|
1186
|
+
ctx: ExtensionContext,
|
|
1187
|
+
owner: ExecState,
|
|
1188
|
+
round: InFlightReview,
|
|
1189
|
+
result: AuditRoundResult,
|
|
1190
|
+
pendingIds: string[],
|
|
1191
|
+
): Promise<void> {
|
|
1192
|
+
// Identity guard: a restore, stop, or fresh handoff replaced the run —
|
|
1193
|
+
// drop this outcome silently (CF2-002).
|
|
1194
|
+
if (execution !== owner) return;
|
|
1195
|
+
if (owner.review.inFlight !== round) return;
|
|
1196
|
+
owner.review.inFlight = null;
|
|
1197
|
+
owner.audit.running = false;
|
|
1198
|
+
if (result !== null && typeof result === "object" && "cancelled" in result) {
|
|
1199
|
+
// Aborted by shutdown/stop/restore/tree-switch: no round, no budget, no wake.
|
|
1200
|
+
updateStatusWidget(ctx);
|
|
1201
|
+
return;
|
|
1202
|
+
}
|
|
1203
|
+
const outcome = result;
|
|
1204
|
+
const reportText = outcome?.report ?? "(review subagent failed to run)";
|
|
1205
|
+
const fingerprintNow = captureReviewFingerprint(ctx, owner);
|
|
1206
|
+
const coveredTaskIds = flattenTaskViews(owner.tasks).map((task) => task.id);
|
|
1207
|
+
if (fingerprintNow !== round.fingerprint) {
|
|
1208
|
+
// The audited subject moved under the reviewer (Q-C): discard, re-run
|
|
1209
|
+
// without budget burn; two consecutive discards commit as undeterminable
|
|
1210
|
+
// so a mutating user cannot loop the loop for free (Q-discard-bound).
|
|
1211
|
+
owner.review.consecutiveDiscards += 1;
|
|
1212
|
+
const runDir = runDirOf(ctx);
|
|
1213
|
+
if (runDir) {
|
|
1214
|
+
writeReviewRoundReport(runDir, {
|
|
1215
|
+
budgetRound: round.budgetRound,
|
|
1216
|
+
attempt: round.attempt,
|
|
1217
|
+
outcome: "discarded",
|
|
1218
|
+
passed: [],
|
|
1219
|
+
failed: [],
|
|
1220
|
+
undeterminable: pendingIds,
|
|
1221
|
+
// v0.9.1 (F-014): the discard report must not understate a
|
|
1222
|
+
// round whose embedded report carries highs.
|
|
1223
|
+
findings: owner.audit.findings,
|
|
1224
|
+
discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
|
|
1225
|
+
fingerprintCaptured: round.fingerprint,
|
|
1226
|
+
fingerprintFound: fingerprintNow,
|
|
1227
|
+
coveredTaskIds,
|
|
1228
|
+
report: reportText,
|
|
1229
|
+
});
|
|
1230
|
+
}
|
|
1231
|
+
if (owner.review.consecutiveDiscards >= 2) {
|
|
1232
|
+
owner.review.consecutiveDiscards = 0;
|
|
1233
|
+
await commitReviewOutcome(
|
|
1234
|
+
ctx,
|
|
1235
|
+
owner,
|
|
1236
|
+
round,
|
|
1237
|
+
{ passed: [], failed: [], undeterminable: pendingIds, report: `(two consecutive fingerprint discards — committed as an undeterminable round)\n\n${reportText}` },
|
|
1238
|
+
pendingIds,
|
|
1239
|
+
);
|
|
1240
|
+
return;
|
|
1241
|
+
}
|
|
1242
|
+
persist(ctx);
|
|
1243
|
+
updateStatusWidget(ctx);
|
|
1244
|
+
await maybeContinueReview(ctx, owner);
|
|
1245
|
+
return;
|
|
1246
|
+
}
|
|
1247
|
+
owner.review.consecutiveDiscards = 0;
|
|
1248
|
+
await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
|
|
1249
|
+
}
|
|
1250
|
+
|
|
1251
|
+
/** v0.9 (findings-driven fix loop): append plan tasks for high findings no
|
|
1252
|
+
* existing task owns. The reviewer stays read-only — it proposes the title
|
|
1253
|
+
* (proposed-task); this machinery applies it with provenance, so every high
|
|
1254
|
+
* finding always has an owner the executor can close, and the stable F-###
|
|
1255
|
+
* id rides in the task title for traceability. Best-effort: an unwritable
|
|
1256
|
+
* plan must not crash the loop (the finding then stays stranded and the cap
|
|
1257
|
+
* pause surfaces it). Returns the appended task ids. */
|
|
1258
|
+
function appendFindingTasks(ex: ExecState, highs: ReviewFinding[]): string[] {
|
|
1259
|
+
const appended: string[] = [];
|
|
1260
|
+
let next = flattenTaskViews(ex.tasks).reduce((max, t) => {
|
|
1261
|
+
const m = /^Task-(\d+)$/.exec(t.id);
|
|
1262
|
+
return m ? Math.max(max, Number(m[1])) : max;
|
|
1263
|
+
}, 0);
|
|
1264
|
+
const wave = maxWave(ex.tasks) + 1;
|
|
1265
|
+
try {
|
|
1266
|
+
const planText = fs.readFileSync(ex.planPath, "utf8");
|
|
1267
|
+
const lines = planText.split("\n");
|
|
1268
|
+
// Insert inside the Tasks section: before the Execution Waves
|
|
1269
|
+
// subsection when present, else before the FIRST checklist header the
|
|
1270
|
+
// plan uses (v0.9.1 F-013: legacy plans say `## Verifier Checklist`,
|
|
1271
|
+
// and an EOF fallback would land outside every parsed section), else
|
|
1272
|
+
// at EOF; walk back over blank separators so the bullet lands
|
|
1273
|
+
// adjacent to its siblings.
|
|
1274
|
+
let insertAt = lines.length;
|
|
1275
|
+
const wavesIdx = lines.findIndex((l) => /^###\s+Execution Waves/.test(l));
|
|
1276
|
+
const checklistIdx = lines.findIndex((l) => CHECKLIST_HEADERS.some((h) => new RegExp(`^##\\s+${h}`).test(l)));
|
|
1277
|
+
if (wavesIdx !== -1) insertAt = wavesIdx;
|
|
1278
|
+
else if (checklistIdx !== -1) insertAt = checklistIdx;
|
|
1279
|
+
while (insertAt > 0 && lines[insertAt - 1].trim() === "") insertAt--;
|
|
1280
|
+
// v0.9.1 (F-012): reviewer text becomes task-title metadata at parse
|
|
1281
|
+
// time — strip the microsyntax metacharacters (em/en dashes, `--`
|
|
1282
|
+
// separators, the `;` field delimiter) so an embedded token can never
|
|
1283
|
+
// split title from tail or forge fields.
|
|
1284
|
+
const sanitize = (text: string): string => text.replace(/[—–]/g, "-").replace(/-{2,}/g, "-").replace(/;/g, ",");
|
|
1285
|
+
const entries: Array<{ id: string; title: string }> = [];
|
|
1286
|
+
const newLines: string[] = [];
|
|
1287
|
+
for (const h of highs) {
|
|
1288
|
+
next += 1;
|
|
1289
|
+
const id = `Task-${next}`;
|
|
1290
|
+
const title = `fix ${h.id}: ${sanitize(h.proposedTask ?? h.note ?? "address the finding")} (appended by execution review round ${ex.audit.rounds})`;
|
|
1291
|
+
// v0.9.1 (F-006): carry the wave in the bullet tail so a re-parse
|
|
1292
|
+
// restores the same wave the live tree assigned — without it the
|
|
1293
|
+
// appended remediation task fell back to wave 1 on restore and
|
|
1294
|
+
// hijacked the ▸ anchor.
|
|
1295
|
+
newLines.push(`- \`${id}\`: ${title} — wave: ${wave}`);
|
|
1296
|
+
entries.push({ id, title });
|
|
1297
|
+
}
|
|
1298
|
+
lines.splice(insertAt, 0, ...newLines);
|
|
1299
|
+
fs.writeFileSync(ex.planPath, lines.join("\n"), "utf8");
|
|
1300
|
+
// v0.9.1 (F-008): only a successful plan write mints the live tasks —
|
|
1301
|
+
// pushing before the write left checkpoint entries the plan file does
|
|
1302
|
+
// not contain whenever the write failed.
|
|
1303
|
+
for (const entry of entries) {
|
|
1304
|
+
ex.tasks.push({ id: entry.id, title: entry.title, wave, deps: [], files: [], status: "pending", children: [] });
|
|
1305
|
+
appended.push(entry.id);
|
|
1306
|
+
}
|
|
1307
|
+
} catch {
|
|
1308
|
+
/* best-effort: stranded highs surface via the cap pause */
|
|
1309
|
+
}
|
|
1310
|
+
return appended;
|
|
1311
|
+
}
|
|
1312
|
+
|
|
1313
|
+
async function commitReviewOutcome(
|
|
1314
|
+
ctx: ExtensionContext,
|
|
1315
|
+
ex: ExecState,
|
|
1316
|
+
round: InFlightReview,
|
|
1317
|
+
outcome: AuditOutcome | null,
|
|
1318
|
+
pendingIds: string[],
|
|
1319
|
+
): Promise<void> {
|
|
1320
|
+
// The budget is charged only when an outcome commits — never on discard
|
|
1321
|
+
// or cancellation (CF2-003).
|
|
1322
|
+
ex.audit.rounds = round.budgetRound;
|
|
1323
|
+
const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
|
|
1324
|
+
const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
|
|
1325
|
+
// Anything the round neither passed nor failed is undeterminable: the
|
|
1326
|
+
// report omitted the check, spelled the verdict unreadably, or the
|
|
1327
|
+
// subagent never ran.
|
|
1328
|
+
const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
|
|
1329
|
+
// v0.9: the newest round's reported findings ARE the unresolved set
|
|
1330
|
+
// (stable ids — a problem is resolved only by no longer being reported).
|
|
1331
|
+
// v0.9.1 (F-001): only a REAL parsed report is authoritative. A round that
|
|
1332
|
+
// produced no report at all — spawn failure (outcome === null) or the
|
|
1333
|
+
// two-consecutive-discard synthesis — must PRESERVE the unresolved set:
|
|
1334
|
+
// clearing it let the vacuous completion guard (empty pendingIds) mark a
|
|
1335
|
+
// run done with its high finding silently dropped.
|
|
1336
|
+
const reported = outcome !== null && outcome.findings !== undefined;
|
|
1337
|
+
const findings = reported ? outcome.findings : ex.audit.findings;
|
|
1338
|
+
ex.audit.findings = findings;
|
|
1339
|
+
const highs = findings.filter((f) => f.severity === "high");
|
|
1340
|
+
// Findings are actionable only when this round actually reported them;
|
|
1341
|
+
// see the fix-loop branch below (v0.9.1, F-001).
|
|
1342
|
+
const actionableHighs = reported ? highs : [];
|
|
1343
|
+
const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
|
|
1344
|
+
const reportText = outcome?.report ?? "(review subagent failed to run)";
|
|
1345
|
+
let reportPath: string | null = null;
|
|
1346
|
+
{
|
|
1347
|
+
const runDir = runDirOf(ctx);
|
|
1348
|
+
if (runDir) {
|
|
1349
|
+
reportPath = writeReviewRoundReport(runDir, {
|
|
1350
|
+
budgetRound: round.budgetRound,
|
|
1351
|
+
attempt: round.attempt,
|
|
1352
|
+
outcome: outcome === null
|
|
1353
|
+
? "spawn-failed"
|
|
1354
|
+
: failed.length > 0 || actionableHighs.length > 0
|
|
1355
|
+
? "failed"
|
|
1356
|
+
: undeterminable.length > 0 ? "undeterminable" : "passed",
|
|
1357
|
+
passed,
|
|
1358
|
+
failed,
|
|
1359
|
+
undeterminable,
|
|
1360
|
+
findings,
|
|
1361
|
+
fingerprintCaptured: round.fingerprint,
|
|
1362
|
+
coveredTaskIds,
|
|
1363
|
+
report: reportText,
|
|
1364
|
+
});
|
|
1365
|
+
}
|
|
1366
|
+
}
|
|
1367
|
+
// Partial progress counts: a check affirmed this round is done even when
|
|
1368
|
+
// a sibling failed, so a later round only re-judges what is still open.
|
|
1369
|
+
for (const id of passed) {
|
|
1370
|
+
const item = ex.items.find((candidate) => candidate.id === id);
|
|
1371
|
+
if (item) item.done = true;
|
|
1372
|
+
}
|
|
1373
|
+
|
|
1374
|
+
// v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
|
|
1375
|
+
// high finding can never complete the run (liveness). Failed checks and
|
|
1376
|
+
// high findings drive ONE union rollback and exactly one executor wake;
|
|
1377
|
+
// high wins over the undeterminable self-schedule (a finding is
|
|
1378
|
+
// actionable independent of verdict evidence). v0.9.1 (F-001): verdicts
|
|
1379
|
+
// are authoritative whenever a report exists, but the FINDINGS-driven
|
|
1380
|
+
// half of the branch needs a findings-bearing report — a no-findings
|
|
1381
|
+
// round (spawn failure, discard synthesis, legacy shape) preserves the
|
|
1382
|
+
// unresolved set and self-schedules instead of rolling back on it.
|
|
1383
|
+
if (failed.length > 0 || actionableHighs.length > 0) {
|
|
1384
|
+
ex.audit.failed = failed;
|
|
1385
|
+
ex.audit.undeterminable = undeterminable;
|
|
1386
|
+
const rolledBack: string[] = [];
|
|
1387
|
+
for (const id of failed) {
|
|
1388
|
+
rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
|
|
1389
|
+
}
|
|
1390
|
+
// Invalidation is the VC-fail invariant ONLY: a check re-verifies its
|
|
1391
|
+
// reopened tasks when a FAILED check rolled them back. A pure
|
|
1392
|
+
// finding-driven rollback keeps earlier passes — the findings channel
|
|
1393
|
+
// itself re-examines the repaired work next round (stable ids).
|
|
1394
|
+
if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
|
|
1395
|
+
const knownIds = new Set(coveredTaskIds);
|
|
1396
|
+
const mappedHighIds = [...new Set(actionableHighs.flatMap((h) => h.taskIds).filter((id) => knownIds.has(id)))];
|
|
1397
|
+
const highRolledBack = findingsRollbackSet(ex.tasks, mappedHighIds);
|
|
1398
|
+
const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
|
|
1399
|
+
const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
|
|
1400
|
+
const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
|
|
1401
|
+
withExecutionCheckpoint(ctx, (cp) => {
|
|
1402
|
+
// v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
|
|
1403
|
+
// re-stamp the checkpoint's plan identity in the same revision so a
|
|
1404
|
+
// later /resume-plans accepts the amended plan instead of rejecting
|
|
1405
|
+
// it as plan-mismatch (which would cost a full re-approval).
|
|
1406
|
+
const amendedCp = amended.length > 0
|
|
1407
|
+
? applyExecutionPlanAmended(cp, planIdentityOf(ex.planPath, cp.plan?.version ?? 1), ex.audit.rounds)
|
|
1408
|
+
: cp;
|
|
1409
|
+
return applyExecutionProgress(amendedCp, {
|
|
1410
|
+
tasks: taskProgressMap(ex.tasks),
|
|
1411
|
+
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1412
|
+
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
|
|
852
1413
|
});
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
1414
|
+
});
|
|
1415
|
+
if (allRolledBack.length > 0 || amended.length > 0) {
|
|
1416
|
+
// Rolling back (or appending) is itself forward progress for the
|
|
1417
|
+
// watchdog, but NOT for audit.rounds: that counter stays monotonic
|
|
1418
|
+
// so the repair loop is bounded. The repair belongs to the
|
|
1419
|
+
// executor — back to executing.
|
|
1420
|
+
ex.stall.rounds = 0;
|
|
1421
|
+
ex.stall.lastSnapshot = stallSnapshot();
|
|
1422
|
+
setRunStatusForReview(ctx, "executing");
|
|
1423
|
+
}
|
|
1424
|
+
persist(ctx);
|
|
1425
|
+
updateStatusWidget(ctx);
|
|
1426
|
+
// v0.8 wake, generalized (v0.9): per-round one-shot token — exactly one
|
|
1427
|
+
// triggerTurn per committed fix-needing outcome, even when the round
|
|
1428
|
+
// resolves long after the settle that spawned it.
|
|
1429
|
+
if (!round.wakeSent) {
|
|
1430
|
+
round.wakeSent = true;
|
|
1431
|
+
const openTasks = flattenTaskViews(ex.tasks)
|
|
1432
|
+
.filter((task) => !taskIsTerminal(task))
|
|
1433
|
+
.map((task) => task.id);
|
|
1434
|
+
const stranded = allRolledBack.length === 0 && amended.length === 0;
|
|
1435
|
+
const highLines = actionableHighs.map((h) => `- ${h.id}${h.taskIds.length ? ` (${h.taskIds.join(", ")})` : ""}: ${h.note}`).join("\n");
|
|
1436
|
+
const reportRef = reportPath
|
|
1437
|
+
? `Full round report: ${reportPath}`
|
|
1438
|
+
: `Full round report (run dir unwritable — inline):\n\n---\n${reportText.slice(0, 4000)}`;
|
|
1439
|
+
// v0.9.1 (F-004): a pure VC-fail round keeps the v0.8 lead — never
|
|
1440
|
+
// announce "0 high-severity findings" over an empty block.
|
|
1441
|
+
const findingsLead = actionableHighs.length > 0
|
|
1442
|
+
? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
|
|
1443
|
+
: `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
|
|
1444
|
+
const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
|
|
1445
|
+
const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
|
|
1446
|
+
messaging().sendMessage(
|
|
1447
|
+
{
|
|
1448
|
+
customType: "pi-plans-audit-failed",
|
|
1449
|
+
content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}`,
|
|
1450
|
+
display: true,
|
|
1451
|
+
},
|
|
1452
|
+
{ triggerTurn: true },
|
|
1453
|
+
);
|
|
1454
|
+
}
|
|
1455
|
+
return; // The agent repairs; the next settle re-enters the loop.
|
|
1456
|
+
}
|
|
1457
|
+
|
|
1458
|
+
// Fail-closed completion: every pending check affirmatively passed AND no
|
|
1459
|
+
// high finding remains. The highs guard is explicit (v0.9.1, F-001): a
|
|
1460
|
+
// no-report round skips the fix branch above, so this is the last line
|
|
1461
|
+
// against vacuously completing an empty-pendingIds round with an
|
|
1462
|
+
// unresolved high. An all-undeterminable round yields failed === [] —
|
|
1463
|
+
// completing here would be fail-open, marking a run done with nothing
|
|
1464
|
+
// verified.
|
|
1465
|
+
if (passed.length === pendingIds.length && highs.length === 0) {
|
|
862
1466
|
ex.audit.failed = [];
|
|
1467
|
+
ex.audit.undeterminable = [];
|
|
863
1468
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
864
1469
|
applyExecutionProgress(cp, {
|
|
865
1470
|
tasks: taskProgressMap(ex.tasks),
|
|
866
1471
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
867
|
-
audit: { rounds: ex.audit.rounds, passed: true },
|
|
1472
|
+
audit: { rounds: ex.audit.rounds, passed: true, findings },
|
|
868
1473
|
}),
|
|
869
1474
|
);
|
|
870
1475
|
await completeExecution(ctx);
|
|
871
1476
|
return;
|
|
872
1477
|
}
|
|
873
|
-
|
|
874
|
-
//
|
|
875
|
-
//
|
|
876
|
-
for (const id of failed) {
|
|
877
|
-
const item = ex.items.find((candidate) => candidate.id === id);
|
|
878
|
-
if (item) item.done = false;
|
|
879
|
-
}
|
|
1478
|
+
|
|
1479
|
+
// Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
|
|
1480
|
+
// wake, no message; the dashboard/overlay carries the round counter.
|
|
880
1481
|
ex.audit.failed = failed;
|
|
881
|
-
|
|
882
|
-
for (const id of failed) {
|
|
883
|
-
rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
|
|
884
|
-
}
|
|
1482
|
+
ex.audit.undeterminable = undeterminable;
|
|
885
1483
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
886
1484
|
applyExecutionProgress(cp, {
|
|
887
1485
|
tasks: taskProgressMap(ex.tasks),
|
|
888
1486
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
889
|
-
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
|
|
1487
|
+
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
|
|
890
1488
|
}),
|
|
891
1489
|
);
|
|
892
|
-
ex.stall.rounds = 0;
|
|
893
|
-
ex.stall.lastSnapshot = stallSnapshot();
|
|
894
1490
|
persist(ctx);
|
|
895
1491
|
updateStatusWidget(ctx);
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
1492
|
+
await maybeContinueReview(ctx, ex);
|
|
1493
|
+
}
|
|
1494
|
+
|
|
1495
|
+
async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promise<void> {
|
|
1496
|
+
if (execution !== ex) return;
|
|
1497
|
+
if (!reviewOwed(ex)) return;
|
|
1498
|
+
if (ex.stall.paused) return;
|
|
1499
|
+
if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
|
|
1500
|
+
pauseReviewCap(ctx, ex);
|
|
1501
|
+
return;
|
|
1502
|
+
}
|
|
1503
|
+
await startReviewRound(ctx);
|
|
1504
|
+
}
|
|
1505
|
+
|
|
1506
|
+
/** Launch a review round under the mode rule: detached (fire-and-forget with
|
|
1507
|
+
* the chain tracked for the test seam) in tui/rpc; awaited inline otherwise
|
|
1508
|
+
* (print/json settle must hold the runtime open through the round). Returns
|
|
1509
|
+
* the chain when the caller must await it, null when detached. */
|
|
1510
|
+
function launchReviewRound(ctx: ExtensionContext): Promise<void> | null {
|
|
1511
|
+
const detach = ctx.mode === "tui" || ctx.mode === "rpc";
|
|
1512
|
+
const chain = (async () => {
|
|
1513
|
+
await startReviewRound(ctx);
|
|
1514
|
+
})();
|
|
1515
|
+
activeReviewChain = chain;
|
|
1516
|
+
if (detach) {
|
|
1517
|
+
chain.catch(() => {
|
|
1518
|
+
/* surfaced via the review messages */
|
|
1519
|
+
});
|
|
1520
|
+
return null;
|
|
1521
|
+
}
|
|
1522
|
+
return chain;
|
|
1523
|
+
}
|
|
1524
|
+
|
|
1525
|
+
/** Deterministic seam: await the in-flight (or self-scheduling) review chain. */
|
|
1526
|
+
export function __awaitReviewRoundForTests(): Promise<void> {
|
|
1527
|
+
return activeReviewChain ?? Promise.resolve();
|
|
1528
|
+
}
|
|
1529
|
+
|
|
1530
|
+
/** When this module graph was first imported into the running pi process.
|
|
1531
|
+
* pi loads extensions once, so a fix written to disk mid-session stays
|
|
1532
|
+
* invisible until /reload — which is exactly why an unreadable audit verdict
|
|
1533
|
+
* deserves a /reload hint rather than a bare retry. */
|
|
1534
|
+
const extensionModuleLoadedAt = new Date();
|
|
1535
|
+
|
|
1536
|
+
/** The /reload advice when the extension on disk is newer than the copy this
|
|
1537
|
+
* process loaded, else null. Shared with /plans via src/staleness.ts so both
|
|
1538
|
+
* report the same answer from one probe. */
|
|
1539
|
+
function staleReloadHint(): string | null {
|
|
1540
|
+
try {
|
|
1541
|
+
const root = path.dirname(path.dirname(new URL(import.meta.url).pathname));
|
|
1542
|
+
return probeStaleReload(root, extensionModuleLoadedAt);
|
|
1543
|
+
} catch {
|
|
1544
|
+
return null;
|
|
1545
|
+
}
|
|
915
1546
|
}
|
|
916
1547
|
|
|
917
1548
|
/** True when the session can surface a pause to a human (D-022): interactive
|
|
@@ -931,7 +1562,7 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
|
|
|
931
1562
|
const run = getRun(ctx.cwd, active.run_id);
|
|
932
1563
|
if (!run) return null;
|
|
933
1564
|
if (phase === "planning" && run.status !== "planning") return null;
|
|
934
|
-
if (phase === "execution" && run.status !== "executing") return null;
|
|
1565
|
+
if (phase === "execution" && run.status !== "executing" && run.status !== "verifying") return null;
|
|
935
1566
|
scaffoldVccSettings(stateRoot);
|
|
936
1567
|
return { settings: loadVccSettings(stateRoot), runId: run.run_id, artifactDir: run.artifact_dir };
|
|
937
1568
|
}
|
|
@@ -1392,6 +2023,9 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
|
|
|
1392
2023
|
|
|
1393
2024
|
export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
|
|
1394
2025
|
if (!execution) return;
|
|
2026
|
+
// A stopped run's in-flight review round dies with it (typed cancelled —
|
|
2027
|
+
// no budget, no wake).
|
|
2028
|
+
abortInFlightReview();
|
|
1395
2029
|
resetExecutionCompactionState(ctx);
|
|
1396
2030
|
pendingExecutionFlush = false;
|
|
1397
2031
|
persist(ctx);
|
|
@@ -1437,7 +2071,7 @@ function stallSnapshot(): string {
|
|
|
1437
2071
|
}
|
|
1438
2072
|
|
|
1439
2073
|
/**
|
|
1440
|
-
* v0.7.1: shared "the
|
|
2074
|
+
* v0.7.1: shared "the execution review is owed" predicate. Every entry point
|
|
1441
2075
|
* that can start the audit (turn_end, agent_before_settle, restoreFromSession,
|
|
1442
2076
|
* the resume path) goes through this so they can never disagree.
|
|
1443
2077
|
*
|
|
@@ -1446,12 +2080,13 @@ function stallSnapshot(): string {
|
|
|
1446
2080
|
*/
|
|
1447
2081
|
function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
|
|
1448
2082
|
if (!ex) return false;
|
|
1449
|
-
// A paused run is never self-driven: the stall /
|
|
2083
|
+
// A paused run is never self-driven: the stall / review-cap pause is an
|
|
1450
2084
|
// explicit "hand control back" signal, and resuming it is the user's call.
|
|
1451
2085
|
// This also bounds the zero-input continue loop in agent_before_settle.
|
|
1452
2086
|
if (ex.stall.paused) return false;
|
|
1453
|
-
if (ex.
|
|
1454
|
-
return allTasksTerminal(ex.tasks)
|
|
2087
|
+
if (ex.review.inFlight) return false;
|
|
2088
|
+
return allTasksTerminal(ex.tasks)
|
|
2089
|
+
&& (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
|
|
1455
2090
|
}
|
|
1456
2091
|
|
|
1457
2092
|
/** v0.7.1: record that this settle already ran (or declined) its audit, so a
|
|
@@ -1564,48 +2199,78 @@ export function filterContinuationMessages<T extends { customType?: string; deta
|
|
|
1564
2199
|
return filterGoalWaitMessages(messages);
|
|
1565
2200
|
}
|
|
1566
2201
|
|
|
1567
|
-
/** Prefix of the stall reason used for the audit-cap pause (D-022). */
|
|
1568
|
-
const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
1569
|
-
|
|
1570
2202
|
/** Called for genuine user input or an explicit same-execution resume.
|
|
1571
|
-
*
|
|
1572
|
-
*
|
|
1573
|
-
*
|
|
2203
|
+
* v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
|
|
2204
|
+
* refill the five-round budget (CF2-004); only /plans-execute
|
|
2205
|
+
* (resumeActiveExecution) is the explicit confirmation surface. Genuine
|
|
2206
|
+
* stall pauses still clear on input as before. */
|
|
1574
2207
|
export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
1575
2208
|
const ex = getExecution();
|
|
1576
2209
|
if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
|
|
1577
|
-
|
|
2210
|
+
if (isReviewCapPause(ex.stall.pausedReason)) {
|
|
2211
|
+
// Surfaced once per input so the user is not left guessing why the run
|
|
2212
|
+
// stays paused; the pause itself and the budget survive untouched.
|
|
2213
|
+
ctx.ui.notify?.(
|
|
2214
|
+
"pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
|
|
2215
|
+
"warning",
|
|
2216
|
+
);
|
|
2217
|
+
return false;
|
|
2218
|
+
}
|
|
1578
2219
|
ex.stall.paused = false;
|
|
1579
2220
|
ex.stall.pausedReason = undefined;
|
|
1580
2221
|
ex.stall.rounds = 0;
|
|
1581
2222
|
ex.stall.lastSnapshot = stallSnapshot();
|
|
1582
|
-
|
|
1583
|
-
ex.audit.rounds = 0;
|
|
1584
|
-
ex.audit.failed = [];
|
|
1585
|
-
withExecutionCheckpoint(ctx, (cp) =>
|
|
1586
|
-
applyExecutionProgress(cp, {
|
|
1587
|
-
tasks: taskProgressMap(ex.tasks),
|
|
1588
|
-
audit: { rounds: 0, lastResult: undefined },
|
|
1589
|
-
pausedReason: null,
|
|
1590
|
-
}),
|
|
1591
|
-
);
|
|
1592
|
-
} else {
|
|
1593
|
-
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
|
|
1594
|
-
}
|
|
2223
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
|
|
1595
2224
|
persist(ctx);
|
|
1596
2225
|
updateStatusWidget(ctx);
|
|
1597
2226
|
return true;
|
|
1598
2227
|
}
|
|
1599
2228
|
|
|
1600
2229
|
export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
2230
|
+
// v0.8: /plans-execute is THE explicit confirmation surface for a
|
|
2231
|
+
// review-cap pause — the only place a fresh five-round budget is granted
|
|
2232
|
+
// (Q-confirm-surface). Ordinary input and session restores never refill.
|
|
2233
|
+
const pausedEx = getExecution();
|
|
2234
|
+
if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
|
|
2235
|
+
pausedEx.stall.paused = false;
|
|
2236
|
+
pausedEx.stall.pausedReason = undefined;
|
|
2237
|
+
pausedEx.stall.rounds = 0;
|
|
2238
|
+
pausedEx.stall.lastSnapshot = stallSnapshot();
|
|
2239
|
+
pausedEx.audit.rounds = 0;
|
|
2240
|
+
pausedEx.audit.failed = [];
|
|
2241
|
+
pausedEx.audit.undeterminable = [];
|
|
2242
|
+
// v0.9: the fresh budget inherits unresolved findings (stable ids keep
|
|
2243
|
+
// counting) — only the round counter resets.
|
|
2244
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
2245
|
+
applyExecutionProgress(cp, {
|
|
2246
|
+
tasks: taskProgressMap(pausedEx.tasks),
|
|
2247
|
+
audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
|
|
2248
|
+
pausedReason: null,
|
|
2249
|
+
}),
|
|
2250
|
+
);
|
|
2251
|
+
persist(ctx);
|
|
2252
|
+
updateStatusWidget(ctx);
|
|
2253
|
+
messaging().sendMessage(
|
|
2254
|
+
{
|
|
2255
|
+
customType: "pi-plans-review-budget-granted",
|
|
2256
|
+
content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
|
|
2257
|
+
display: true,
|
|
2258
|
+
},
|
|
2259
|
+
{ triggerTurn: false },
|
|
2260
|
+
);
|
|
2261
|
+
const grantChain = launchReviewRound(ctx);
|
|
2262
|
+
if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
|
|
2263
|
+
return true;
|
|
2264
|
+
}
|
|
1601
2265
|
// v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
|
|
1602
2266
|
// to `return false` here, so `/plans-execute` answered "already executing"
|
|
1603
2267
|
// and the run stayed stranded until a full re-entry or a session restore.
|
|
1604
2268
|
// It is not paused, so the pause path below cannot see it — check it first
|
|
1605
|
-
// and run the owed
|
|
2269
|
+
// and run the owed review instead of reporting "nothing to resume".
|
|
1606
2270
|
if (!execution?.stall.paused && pendingAudit()) {
|
|
1607
2271
|
latchAuditThisSettle();
|
|
1608
|
-
|
|
2272
|
+
const chain = launchReviewRound(ctx);
|
|
2273
|
+
if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
|
|
1609
2274
|
return true;
|
|
1610
2275
|
}
|
|
1611
2276
|
if (!resumeGoalWaitIfPaused(ctx)) return false;
|
|
@@ -1626,6 +2291,12 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
1626
2291
|
const summary = flat
|
|
1627
2292
|
.map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
|
|
1628
2293
|
.join("\n");
|
|
2294
|
+
// v0.9: residual (non-high) findings are summarized, never silent — the
|
|
2295
|
+
// loop converged because nothing high-blocking remained.
|
|
2296
|
+
const residualFindings = execution.audit.findings.filter((f) => f.severity !== "high");
|
|
2297
|
+
const residualNote = residualFindings.length > 0
|
|
2298
|
+
? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
|
|
2299
|
+
: "";
|
|
1629
2300
|
const planPath = execution.planPath;
|
|
1630
2301
|
withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
|
|
1631
2302
|
execution = null;
|
|
@@ -1635,7 +2306,7 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
1635
2306
|
messaging().sendMessage(
|
|
1636
2307
|
{
|
|
1637
2308
|
customType: "pi-plans-complete",
|
|
1638
|
-
content: `**Plan complete!** ✅ \`${planPath}\` —
|
|
2309
|
+
content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
|
|
1639
2310
|
display: true,
|
|
1640
2311
|
},
|
|
1641
2312
|
{ triggerTurn: false },
|
|
@@ -1668,7 +2339,11 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
|
|
|
1668
2339
|
? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
|
|
1669
2340
|
: graphBlockForExecutor(mode === "enabled");
|
|
1670
2341
|
const rollbackNote = execution.audit.failed.length > 0
|
|
1671
|
-
? `\
|
|
2342
|
+
? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
|
|
2343
|
+
: "";
|
|
2344
|
+
const unresolvedHighs = unresolvedHighFindings(execution);
|
|
2345
|
+
const highFindingsNote = unresolvedHighs.length > 0
|
|
2346
|
+
? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
|
|
1672
2347
|
: "";
|
|
1673
2348
|
return `[PI-PLANS EXECUTION — write access enabled]
|
|
1674
2349
|
Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
|
|
@@ -1676,7 +2351,7 @@ Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${p
|
|
|
1676
2351
|
Current wave ${currentWave} open tasks:
|
|
1677
2352
|
${waveList}
|
|
1678
2353
|
|
|
1679
|
-
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
|
|
2354
|
+
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}
|
|
1680
2355
|
|
|
1681
2356
|
${graphLine}
|
|
1682
2357
|
|
|
@@ -1684,7 +2359,7 @@ Execution rules:
|
|
|
1684
2359
|
- Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
|
|
1685
2360
|
- Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
|
|
1686
2361
|
- Close subtasks before their parent; a parent is auditable only when every child is terminal.
|
|
1687
|
-
- When every task is terminal, the independent
|
|
2362
|
+
- When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
|
|
1688
2363
|
- Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
|
|
1689
2364
|
- Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
|
|
1690
2365
|
- Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
|
|
@@ -1741,10 +2416,18 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1741
2416
|
planTasks = parsePlanTasks(planText);
|
|
1742
2417
|
const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
|
|
1743
2418
|
tasks = buildTaskView(planTasks, snapshotProgress);
|
|
1744
|
-
|
|
2419
|
+
// Fresh text wins, but the satisfied state survives the re-parse: a
|
|
2420
|
+
// mid-loop session restore (v0.9.1, found via F-001's self-schedule
|
|
2421
|
+
// path) must not hand the next round a brief that re-judges checks an
|
|
2422
|
+
// earlier round already passed — that burned budget on every /reload.
|
|
2423
|
+
const snapDone = new Set(snapshot.items.filter((c) => c.done).map((c) => c.id));
|
|
2424
|
+
items = parseChecklist(planText).map((item) => (snapDone.has(item.id) ? { ...item, done: true } : item));
|
|
1745
2425
|
} catch {
|
|
1746
2426
|
tasks = snapshot.tasks;
|
|
1747
2427
|
}
|
|
2428
|
+
// The snapshot cannot carry an in-flight round (it is memory-only); abort
|
|
2429
|
+
// any live one from the previous session graph and rebuild fresh (CF2-002).
|
|
2430
|
+
abortInFlightReview();
|
|
1748
2431
|
execution = {
|
|
1749
2432
|
planPath: snapshot.planPath,
|
|
1750
2433
|
items,
|
|
@@ -1754,8 +2437,15 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1754
2437
|
startedAt: snapshot.startedAt,
|
|
1755
2438
|
usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
|
|
1756
2439
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
1757
|
-
stall: { ...snapshot.stall, lastSnapshot: null
|
|
1758
|
-
audit: {
|
|
2440
|
+
stall: { ...snapshot.stall, lastSnapshot: null },
|
|
2441
|
+
audit: {
|
|
2442
|
+
rounds: snapshot.audit?.rounds ?? 0,
|
|
2443
|
+
failed: snapshot.audit?.failed ?? [],
|
|
2444
|
+
undeterminable: [],
|
|
2445
|
+
findings: toReviewFindings(snapshot.audit?.findings),
|
|
2446
|
+
running: false,
|
|
2447
|
+
},
|
|
2448
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
1759
2449
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
1760
2450
|
};
|
|
1761
2451
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
@@ -1765,9 +2455,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1765
2455
|
if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
|
|
1766
2456
|
persist(ctx);
|
|
1767
2457
|
if (pendingAudit()) {
|
|
1768
|
-
//
|
|
2458
|
+
// Self-heal (v0.8): a verifying run with pending checks restarts its
|
|
2459
|
+
// round exactly once per resume — the full predicate (paused, in-flight,
|
|
2460
|
+
// budget) lives inside startReviewRound. Restores never grant budget.
|
|
1769
2461
|
latchAuditThisSettle();
|
|
1770
|
-
|
|
2462
|
+
const chain = launchReviewRound(ctx);
|
|
2463
|
+
if (chain) await chain;
|
|
1771
2464
|
}
|
|
1772
2465
|
updateStatusWidget(ctx);
|
|
1773
2466
|
}
|