pi-plans 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +40 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +13 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +157 -56
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +56 -10
- package/src/exec.ts +614 -126
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +13 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +39 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +18 -5
- package/tests/auditor.test.ts +116 -17
- package/tests/dashboard.test.ts +135 -1
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +65 -0
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
package/src/exec.ts
CHANGED
|
@@ -9,15 +9,16 @@
|
|
|
9
9
|
* live progress (compact aboveEditor widget; Ctrl+Shift+T expands the full
|
|
10
10
|
* tree), a stall watchdog pauses the run when consecutive rounds produce no
|
|
11
11
|
* task-state change, and when every task reaches a terminal state an
|
|
12
|
-
* independent
|
|
13
|
-
* failed checks roll their covered tasks
|
|
14
|
-
* channel), and
|
|
15
|
-
*
|
|
12
|
+
* independent execution reviewer verifies the plan's verification checks in
|
|
13
|
+
* a detached, overlay-visible loop — failed checks roll their covered tasks
|
|
14
|
+
* back to pending (audit-flow-only channel), and five committed rounds pause
|
|
15
|
+
* the run for the user in every mode (fail-closed, never a silent stop).
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import * as fs from "node:fs";
|
|
19
19
|
import * as path from "node:path";
|
|
20
|
-
import { randomUUID } from "node:crypto";
|
|
20
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
21
|
+
import { execSync } from "node:child_process";
|
|
21
22
|
import type {
|
|
22
23
|
CompactionResult,
|
|
23
24
|
ExtensionAPI,
|
|
@@ -44,7 +45,7 @@ import {
|
|
|
44
45
|
type VccCompactionBuildResult,
|
|
45
46
|
type VccCompactionStats,
|
|
46
47
|
} from "./compaction.ts";
|
|
47
|
-
import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
48
|
+
import { TERMINAL_RUN_STATUSES, getRun, latestRun, lintPlanIntoNotices, loadConfig, readActive, resolveStateRootOrNull, runDirPath, setRunStatus, StateError, utcNow } from "./state.ts";
|
|
48
49
|
import { resolveUiLanguage, type UiLanguage } from "./ui-language.ts";
|
|
49
50
|
import { bindRun, resolveActiveRun } from "./run-context.ts";
|
|
50
51
|
import type { SubagentProgressEvent } from "./subagent.ts";
|
|
@@ -82,6 +83,7 @@ import {
|
|
|
82
83
|
buildTaskView,
|
|
83
84
|
currentTask,
|
|
84
85
|
flattenTaskViews,
|
|
86
|
+
invalidateChecksForRolledBackTasks,
|
|
85
87
|
taskIsTerminal,
|
|
86
88
|
taskProgress,
|
|
87
89
|
taskProgressMap,
|
|
@@ -97,8 +99,13 @@ import {
|
|
|
97
99
|
renderDashboardLines,
|
|
98
100
|
renderDashboardTreeLines,
|
|
99
101
|
} from "./dashboard.ts";
|
|
100
|
-
import {
|
|
102
|
+
import { REVIEW_MAX_ROUNDS, presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditRoundResult } from "./auditor.ts";
|
|
103
|
+
import { staleReloadHint as probeStaleReload } from "./staleness.ts";
|
|
101
104
|
import { messaging } from "./messaging.ts";
|
|
105
|
+
import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
|
|
106
|
+
import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
|
|
107
|
+
import { resolveReviewerSpawn } from "./thinking-levels.ts";
|
|
108
|
+
import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
|
|
102
109
|
|
|
103
110
|
export interface ExecState {
|
|
104
111
|
planPath: string;
|
|
@@ -117,8 +124,14 @@ export interface ExecState {
|
|
|
117
124
|
/** Stall watchdog (v0.6.1): consecutive settled rounds without a task
|
|
118
125
|
* status change; auto-pause at the cap. */
|
|
119
126
|
stall: { rounds: number; lastSnapshot: string | null; paused: boolean; pausedReason?: string };
|
|
120
|
-
/** Completion-audit bookkeeping.
|
|
121
|
-
|
|
127
|
+
/** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
|
|
128
|
+
* only when a round outcome commits (never on discard/cancel) — and is the
|
|
129
|
+
* only piece persisted. */
|
|
130
|
+
audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
|
|
131
|
+
/** Execution-review loop (v0.8), memory-only: the attempt index names the
|
|
132
|
+
* per-round report files; consecutiveDiscards bounds the fingerprint
|
|
133
|
+
* re-run loop; inFlight owns the round's abort lifecycle. */
|
|
134
|
+
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
|
|
122
135
|
/** Per-settle audit latch (v0.7.1): a settled round fires the completion
|
|
123
136
|
* audit at most once, so the turn_end / agent_before_settle / resume entry
|
|
124
137
|
* points cannot double-consume a round when several land in one settle.
|
|
@@ -126,6 +139,18 @@ export interface ExecState {
|
|
|
126
139
|
auditLatch?: { auditedThisSettle: boolean; activity: number };
|
|
127
140
|
}
|
|
128
141
|
|
|
142
|
+
/** One in-flight review round: owns its abort lifecycle, its fingerprint of
|
|
143
|
+
* the audited subject, and the per-round one-shot wake token (v0.8). */
|
|
144
|
+
interface InFlightReview {
|
|
145
|
+
controller: AbortController;
|
|
146
|
+
/** Budget round this attempt belongs to (audit.rounds + 1 at spawn). */
|
|
147
|
+
budgetRound: number;
|
|
148
|
+
/** Monotonic attempt ordinal; names the round report file. */
|
|
149
|
+
attempt: number;
|
|
150
|
+
fingerprint: string;
|
|
151
|
+
wakeSent: boolean;
|
|
152
|
+
}
|
|
153
|
+
|
|
129
154
|
/**
|
|
130
155
|
* D-008 (issue #3): re-resolve the chrome language and repaint the dashboard
|
|
131
156
|
* and status bar right after a `set-language` change.
|
|
@@ -253,9 +278,12 @@ export function loadExecutionFromCheckpoint(
|
|
|
253
278
|
const reverifyAll = cp.execution.reverifyAll === true || headChanged || headUnverifiable;
|
|
254
279
|
const progress: TaskProgressMap = reverifyAll ? {} : (cp.execution.tasks ?? {});
|
|
255
280
|
const tasks = buildTaskView(planTasks, progress);
|
|
256
|
-
//
|
|
257
|
-
//
|
|
258
|
-
|
|
281
|
+
// v0.8: a review-cap pause SURVIVES the restore — the budget must stay
|
|
282
|
+
// bounded across restarts; only /plans-execute grants a fresh one. The
|
|
283
|
+
// legacy v0.7 prefix stays dual-matched for one release.
|
|
284
|
+
const wasReviewCapPause = isReviewCapPause(cp.execution.pausedReason);
|
|
285
|
+
// Any live round from the replaced session graph dies here (CF2-002).
|
|
286
|
+
abortInFlightReview();
|
|
259
287
|
if (!reverifyAll) {
|
|
260
288
|
for (const id of cp.execution.doneVcIds) {
|
|
261
289
|
const item = items.find((candidate) => candidate.id === id);
|
|
@@ -271,22 +299,24 @@ export function loadExecutionFromCheckpoint(
|
|
|
271
299
|
startedAt: utcNow(),
|
|
272
300
|
usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
|
|
273
301
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
274
|
-
// D-020: a paused legacy (or stopped) execution rebuilds unpaused —
|
|
275
|
-
//
|
|
276
|
-
//
|
|
277
|
-
//
|
|
278
|
-
//
|
|
279
|
-
|
|
280
|
-
rounds: 0,
|
|
302
|
+
// D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
|
|
303
|
+
// resume itself is the user's intent; the reason is surfaced in the
|
|
304
|
+
// resume brief instead. EXCEPT a review-cap pause (v0.8): it must
|
|
305
|
+
// survive restores paused and at its committed round count, or the
|
|
306
|
+
// 5-round budget would never bound anything across restarts.
|
|
307
|
+
stall: {
|
|
308
|
+
rounds: cp.execution.stallRounds ?? 0,
|
|
281
309
|
lastSnapshot: null,
|
|
282
|
-
paused:
|
|
283
|
-
pausedReason: undefined,
|
|
310
|
+
paused: wasReviewCapPause,
|
|
311
|
+
pausedReason: wasReviewCapPause ? (cp.execution.pausedReason ?? undefined) : undefined,
|
|
284
312
|
},
|
|
285
313
|
audit: {
|
|
286
|
-
rounds:
|
|
314
|
+
rounds: cp.execution.audit?.rounds ?? 0,
|
|
287
315
|
failed: [],
|
|
316
|
+
undeterminable: cp.execution.audit?.undeterminable ?? [],
|
|
288
317
|
running: false,
|
|
289
318
|
},
|
|
319
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
290
320
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
291
321
|
};
|
|
292
322
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
@@ -295,11 +325,6 @@ export function loadExecutionFromCheckpoint(
|
|
|
295
325
|
resetContinuationRuntime(ctx);
|
|
296
326
|
pendingExecutionFlush = false;
|
|
297
327
|
resetExecutionCompactionState(ctx);
|
|
298
|
-
if (wasAuditCapPause) {
|
|
299
|
-
withExecutionCheckpoint(ctx, (cp2) =>
|
|
300
|
-
applyExecutionProgress(cp2, { audit: { rounds: 0, lastResult: undefined }, pausedReason: null }),
|
|
301
|
-
);
|
|
302
|
-
}
|
|
303
328
|
// D-020: an orphaned v0.6.0 delegated executor never survives a restart.
|
|
304
329
|
// Its checkpoint delegate marker REFUSES the direct load — the run must
|
|
305
330
|
// re-enter through the execution handoff so the C-006 approval gate
|
|
@@ -475,6 +500,8 @@ function updatePanelWidget(ctx: ExtensionContext): void {
|
|
|
475
500
|
pausedReason: current.stall.pausedReason,
|
|
476
501
|
auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
|
|
477
502
|
auditFailed: current.audit.failed,
|
|
503
|
+
auditUndeterminable: current.audit.undeterminable,
|
|
504
|
+
reviewRunning: current.audit.running === true || current.review.inFlight !== null,
|
|
478
505
|
startedAt: current.startedAt,
|
|
479
506
|
usage: current.usage,
|
|
480
507
|
});
|
|
@@ -497,6 +524,8 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
497
524
|
pausedReason: execution.stall.pausedReason,
|
|
498
525
|
auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
|
|
499
526
|
auditFailed: execution.audit.failed,
|
|
527
|
+
auditUndeterminable: execution.audit.undeterminable,
|
|
528
|
+
reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
|
|
500
529
|
});
|
|
501
530
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
|
|
502
531
|
return;
|
|
@@ -524,6 +553,10 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
524
553
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `⌛ plans: ${active.run_id} (executing)`));
|
|
525
554
|
return;
|
|
526
555
|
}
|
|
556
|
+
if (status === "verifying") {
|
|
557
|
+
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", `🔎 plans: ${active.run_id} (verifying)`));
|
|
558
|
+
return;
|
|
559
|
+
}
|
|
527
560
|
if (status === "planning") {
|
|
528
561
|
const emoji = parseLatestPlanExists(active.artifact_dir) ? "📝" : "💬";
|
|
529
562
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("muted", `${emoji} plans: ${active.run_id}`));
|
|
@@ -594,6 +627,8 @@ export async function startExecution(
|
|
|
594
627
|
ctx: ExtensionContext,
|
|
595
628
|
input: StartExecutionInput,
|
|
596
629
|
): Promise<void> {
|
|
630
|
+
// A fresh handoff replaces any live run — abort its in-flight review round first.
|
|
631
|
+
abortInFlightReview();
|
|
597
632
|
const tasks = buildTaskView(input.planTasks);
|
|
598
633
|
execution = {
|
|
599
634
|
planPath: input.planPath,
|
|
@@ -605,7 +640,8 @@ export async function startExecution(
|
|
|
605
640
|
usage: { inToks: 0, outToks: 0 },
|
|
606
641
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
607
642
|
stall: { rounds: 0, lastSnapshot: null, paused: false },
|
|
608
|
-
audit: { rounds: 0, failed: [], running: false },
|
|
643
|
+
audit: { rounds: 0, failed: [], undeterminable: [], running: false },
|
|
644
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
609
645
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
610
646
|
};
|
|
611
647
|
// Seed the watchdog baseline only after `execution` points at the new state
|
|
@@ -670,9 +706,11 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
|
|
|
670
706
|
applyExecutionProgress(cp, {
|
|
671
707
|
tasks: taskProgressMap(execution!.tasks),
|
|
672
708
|
doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
|
|
709
|
+
stallRounds: execution!.stall.rounds,
|
|
673
710
|
audit: {
|
|
674
711
|
rounds: execution!.audit.rounds,
|
|
675
712
|
lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
|
|
713
|
+
undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
|
|
676
714
|
},
|
|
677
715
|
}),
|
|
678
716
|
);
|
|
@@ -746,15 +784,17 @@ export function registerExecutionTurnHandlers(
|
|
|
746
784
|
maybeContinuationFollowUp(ctx);
|
|
747
785
|
return;
|
|
748
786
|
}
|
|
749
|
-
// Fully settled and still owed
|
|
750
|
-
//
|
|
751
|
-
//
|
|
752
|
-
//
|
|
787
|
+
// Fully settled and still owed a review: launch the round under the
|
|
788
|
+
// mode rule — tui/rpc detach (the settle returns NOW and the overlay
|
|
789
|
+
// carries progress); print/json await inline so runtime teardown cannot
|
|
790
|
+
// kill the child. The round's own outcome routing drives the rest.
|
|
753
791
|
latchAuditThisSettle();
|
|
754
|
-
|
|
792
|
+
const chain = launchReviewRound(ctx);
|
|
793
|
+
if (chain) await chain;
|
|
755
794
|
});
|
|
756
795
|
ext.on("session_shutdown", async (_event, ctx) => {
|
|
757
796
|
drainExecutionFlush(ctx);
|
|
797
|
+
abortInFlightReview();
|
|
758
798
|
execution = null;
|
|
759
799
|
executionRunId = null;
|
|
760
800
|
continuationRuntime = null;
|
|
@@ -795,20 +835,205 @@ export function registerExecutionTurnHandlers(
|
|
|
795
835
|
if (usage) recordExecutionTurn(ctx, usage);
|
|
796
836
|
if (getExecution() && pendingAudit() && !auditLatchOf(execution!).auditedThisSettle) {
|
|
797
837
|
latchAuditThisSettle();
|
|
798
|
-
|
|
838
|
+
const chain = launchReviewRound(ctx);
|
|
839
|
+
if (chain) await chain;
|
|
799
840
|
}
|
|
800
841
|
await onTurnEnd?.(ctx);
|
|
801
842
|
});
|
|
802
843
|
}
|
|
803
844
|
|
|
804
|
-
/**
|
|
805
|
-
* complete the run, keep iterating (rollback), pause at the round cap
|
|
806
|
-
* (interactive), or stop at the cap (auto-approve/headless — D-022).
|
|
845
|
+
/** ==== Execution-review loop (v0.8) ====
|
|
807
846
|
*
|
|
808
|
-
*
|
|
809
|
-
*
|
|
810
|
-
*
|
|
811
|
-
|
|
847
|
+
* When every task is terminal and checks are still owed, the run enters the
|
|
848
|
+
* `verifying` status and a DETACHED read-only reviewer round runs in the
|
|
849
|
+
* background (tui/rpc): the settle handler returns immediately and the
|
|
850
|
+
* executor is truly idle while the overlay shows live progress. In print/json
|
|
851
|
+
* modes the settle handler keeps AWAITING the round inline — runtime
|
|
852
|
+
* teardown at settle would otherwise kill a detached child and swallow the
|
|
853
|
+
* pause signal.
|
|
854
|
+
*
|
|
855
|
+
* Budget: `audit.rounds` counts COMMITTED rounds only (pass, fail, or
|
|
856
|
+
* undeterminable); discards and cancellations burn nothing. Undeterminable
|
|
857
|
+
* rounds self-schedule the retry inside the loop (no wake). Two consecutive
|
|
858
|
+
* fingerprint discards commit as an undeterminable round so the loop stays
|
|
859
|
+
* bounded. Exhaustion pauses in EVERY mode (fail-closed) with an in-band
|
|
860
|
+
* `pi-plans-review-paused` message; the ONLY fresh-budget surface is
|
|
861
|
+
* /plans-execute — ordinary input and session restores never refill.
|
|
862
|
+
*
|
|
863
|
+
* Lifecycle: each round owns a session-scoped AbortController (never
|
|
864
|
+
* ctx.signal, which is turn-scoped), aborted from session_shutdown,
|
|
865
|
+
* stopExecution, startExecution, and restoreFromSession. Outcomes are
|
|
866
|
+
* guarded by execution identity (`execution !== owner` → silent discard).
|
|
867
|
+
*
|
|
868
|
+
* Completion stays fail-closed AND never fail-open: a run completes only
|
|
869
|
+
* when every pending check was affirmatively passed. */
|
|
870
|
+
const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
|
|
871
|
+
/** v0.7 protocol value — dual-matched for one release so checkpoints written
|
|
872
|
+
* by older builds keep their cap pause recognized on restore. */
|
|
873
|
+
const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
874
|
+
|
|
875
|
+
function isReviewCapPause(reason: string | undefined | null): boolean {
|
|
876
|
+
if (!reason) return false;
|
|
877
|
+
return reason.startsWith(REVIEW_CAP_PAUSE_PREFIX) || reason.startsWith(LEGACY_AUDIT_CAP_PAUSE_PREFIX);
|
|
878
|
+
}
|
|
879
|
+
|
|
880
|
+
/** The currently-running (or self-scheduling) review chain; the sanctioned
|
|
881
|
+
* test seam awaits this. */
|
|
882
|
+
let activeReviewChain: Promise<void> | null = null;
|
|
883
|
+
|
|
884
|
+
function abortInFlightReview(): void {
|
|
885
|
+
const inFlight = execution?.review.inFlight;
|
|
886
|
+
if (inFlight) {
|
|
887
|
+
try {
|
|
888
|
+
inFlight.controller.abort();
|
|
889
|
+
} catch {
|
|
890
|
+
/* already aborted */
|
|
891
|
+
}
|
|
892
|
+
if (execution) execution.review.inFlight = null;
|
|
893
|
+
}
|
|
894
|
+
activeReviewChain = null;
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
function reviewOwed(ex: ExecState): boolean {
|
|
898
|
+
return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
|
|
899
|
+
}
|
|
900
|
+
|
|
901
|
+
function runDirOf(ctx: ExtensionContext): string | null {
|
|
902
|
+
const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
|
|
903
|
+
return runId ? runDirPath(ctx.cwd, runId) : null;
|
|
904
|
+
}
|
|
905
|
+
|
|
906
|
+
function setRunStatusForReview(ctx: ExtensionContext, status: "verifying" | "executing"): void {
|
|
907
|
+
const runId = executionRunId ?? resolveActiveRun(ctx.sessionManager, ctx.cwd)?.run_id ?? null;
|
|
908
|
+
if (!runId) return;
|
|
909
|
+
try {
|
|
910
|
+
const current = getRun(ctx.cwd, runId)?.status;
|
|
911
|
+
if (current !== status && current !== "done" && current !== "abandoned") {
|
|
912
|
+
setRunStatus(ctx.cwd, runId, status);
|
|
913
|
+
}
|
|
914
|
+
} catch {
|
|
915
|
+
/* best-effort */
|
|
916
|
+
}
|
|
917
|
+
}
|
|
918
|
+
|
|
919
|
+
/** Round fingerprint (Q-fingerprint-scope): plan digest + git HEAD +
|
|
920
|
+
* covered-file mtimes. A change between round start and resolve means the
|
|
921
|
+
* reviewer judged a subject that no longer exists — discard and re-run. */
|
|
922
|
+
function captureReviewFingerprint(ctx: ExtensionContext, ex: ExecState): string {
|
|
923
|
+
const parts: string[] = [];
|
|
924
|
+
try {
|
|
925
|
+
parts.push(createHash("sha256").update(fs.readFileSync(ex.planPath, "utf8")).digest("hex"));
|
|
926
|
+
} catch {
|
|
927
|
+
parts.push("plan-unreadable");
|
|
928
|
+
}
|
|
929
|
+
try {
|
|
930
|
+
parts.push(execSync("git rev-parse HEAD", { cwd: ctx.cwd, stdio: ["ignore", "pipe", "pipe"] }).toString().trim());
|
|
931
|
+
} catch {
|
|
932
|
+
parts.push("no-head");
|
|
933
|
+
}
|
|
934
|
+
const covered = new Set<string>();
|
|
935
|
+
for (const task of flattenTaskViews(ex.tasks)) {
|
|
936
|
+
for (const file of task.files ?? []) covered.add(file);
|
|
937
|
+
}
|
|
938
|
+
const mtimes: string[] = [];
|
|
939
|
+
for (const file of [...covered].sort()) {
|
|
940
|
+
try {
|
|
941
|
+
mtimes.push(`${file}:${fs.statSync(path.resolve(ctx.cwd, file)).mtimeMs}`);
|
|
942
|
+
} catch {
|
|
943
|
+
mtimes.push(`${file}:missing`);
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
parts.push(mtimes.join("|"));
|
|
947
|
+
return createHash("sha256").update(parts.join("\u0000")).digest("hex");
|
|
948
|
+
}
|
|
949
|
+
|
|
950
|
+
/** Round timeout: a committed round is minutes, never the 60-min subagent
|
|
951
|
+
* default — a hung child must surface as a spawn-failure round, not park the
|
|
952
|
+
* run in verifying for an hour (CF2-010). */
|
|
953
|
+
const REVIEW_ROUND_TIMEOUT_MS = 20 * 60 * 1000;
|
|
954
|
+
|
|
955
|
+
/** Reviewer-role pinning (Q-role-fallback): a CONFIRMED delegated role pins
|
|
956
|
+
* the spawn's model+thinking and labels the overlay with the role; an
|
|
957
|
+
* unconfirmed or current-session role inherits the session default with the
|
|
958
|
+
* "session default" label — a detached round NEVER opens the interactive
|
|
959
|
+
* first-use panel. */
|
|
960
|
+
function reviewSpawnProfile(): { model?: string; thinkingLevel?: string; label: string } {
|
|
961
|
+
try {
|
|
962
|
+
const reviewer = loadGlobalConfig().config.reviewer;
|
|
963
|
+
if (reviewer.mode !== "current-session" && reviewerReady(reviewer)) {
|
|
964
|
+
const spawn = resolveReviewerSpawn(reviewer);
|
|
965
|
+
if (spawn.modelSelector) {
|
|
966
|
+
return { model: spawn.modelSelector, thinkingLevel: spawn.thinkingLevel ?? undefined, label: spawn.label };
|
|
967
|
+
}
|
|
968
|
+
}
|
|
969
|
+
} catch {
|
|
970
|
+
/* fall through to the session default */
|
|
971
|
+
}
|
|
972
|
+
return { label: "session default" };
|
|
973
|
+
}
|
|
974
|
+
|
|
975
|
+
/** Engine-held lane state for the in-flight round: the reopen path builds a
|
|
976
|
+
* fresh one-shot controller seeded from THIS object, so the accumulated
|
|
977
|
+
* transcript survives ESC + reopen (CF2-001 / Q-reopen-seed). */
|
|
978
|
+
let reviewLane: RefineLaneState | null = null;
|
|
979
|
+
let reviewOverlay: RefineOverlayController | null = null;
|
|
980
|
+
let reviewModelLabel: string | undefined;
|
|
981
|
+
|
|
982
|
+
function freshReviewLane(attempt: number): RefineLaneState {
|
|
983
|
+
return {
|
|
984
|
+
id: `review-round-${attempt}`,
|
|
985
|
+
label: `Execution review round ${attempt}`,
|
|
986
|
+
status: "queued",
|
|
987
|
+
phase: "queued",
|
|
988
|
+
detail: "",
|
|
989
|
+
transcript: [],
|
|
990
|
+
currentTurnIndex: 0,
|
|
991
|
+
scrollOffset: 0,
|
|
992
|
+
followTranscript: true,
|
|
993
|
+
viewportHeight: 1,
|
|
994
|
+
};
|
|
995
|
+
}
|
|
996
|
+
|
|
997
|
+
/** Fresh controller per round (the controller is one-shot: closed latch,
|
|
998
|
+
* overlayPromise bail, terminal-lane early return — reuse drops progress).
|
|
999
|
+
* A UI failure must NEVER kill the round itself — best-effort only. */
|
|
1000
|
+
function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
|
|
1001
|
+
if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
|
|
1002
|
+
try {
|
|
1003
|
+
const controller = new RefineOverlayController("auditor", [{ id: lane.id, label: lane.label }], () => {}, lang ?? "en");
|
|
1004
|
+
controller.seedLane(lane);
|
|
1005
|
+
controller.open(refineOverlayContext(ctx), modelLabel);
|
|
1006
|
+
return controller;
|
|
1007
|
+
} catch {
|
|
1008
|
+
return null;
|
|
1009
|
+
}
|
|
1010
|
+
}
|
|
1011
|
+
|
|
1012
|
+
/** The reopen surface (Task-3.4): rebuilds the overlay from engine-held lane
|
|
1013
|
+
* state; inert when no round is in flight. */
|
|
1014
|
+
export function reopenReviewOverlay(ctx: ExtensionContext): void {
|
|
1015
|
+
if (!execution?.review.inFlight || !reviewLane) return;
|
|
1016
|
+
const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
|
|
1017
|
+
if (controller) reviewOverlay = controller;
|
|
1018
|
+
}
|
|
1019
|
+
|
|
1020
|
+
function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
1021
|
+
const detail =
|
|
1022
|
+
ex.audit.failed.length > 0
|
|
1023
|
+
? `failed: ${ex.audit.failed.join(", ")}`
|
|
1024
|
+
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1025
|
+
const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${REVIEW_MAX_ROUNDS} rounds (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh five-round budget; ordinary messages and session restores do not. (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
|
|
1026
|
+
pauseForStall(ctx, reason);
|
|
1027
|
+
// In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
|
|
1028
|
+
// pauseForStall's ui.notify is optional and absent headless, so the pause
|
|
1029
|
+
// must also land in the session stream every mode can read.
|
|
1030
|
+
messaging().sendMessage(
|
|
1031
|
+
{ customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
|
|
1032
|
+
{ triggerTurn: false },
|
|
1033
|
+
);
|
|
1034
|
+
}
|
|
1035
|
+
|
|
1036
|
+
async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
812
1037
|
if (!execution) return;
|
|
813
1038
|
const ex = execution;
|
|
814
1039
|
// Skipped-pass checks resolve without a subagent round.
|
|
@@ -823,43 +1048,194 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
|
|
|
823
1048
|
await completeExecution(ctx);
|
|
824
1049
|
return;
|
|
825
1050
|
}
|
|
826
|
-
if (ex.
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
if (isInteractiveSession(ctx)) {
|
|
830
|
-
pauseForStall(
|
|
831
|
-
ctx,
|
|
832
|
-
`${AUDIT_CAP_PAUSE_PREFIX} ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"}); review the audit reports and fix the failures, then resume with any message or /plans-execute — resuming grants a fresh three-round audit budget (or close the failed checks' tasks as skipped to pass them as skipped-pass)`,
|
|
833
|
-
);
|
|
834
|
-
return;
|
|
835
|
-
}
|
|
836
|
-
await stopExecution(ctx, `completion audit exhausted ${AUDIT_MAX_ROUNDS} rounds (failed: ${ex.audit.failed.join(", ") || "unknown"})`);
|
|
1051
|
+
if (ex.stall.paused || ex.review.inFlight) return;
|
|
1052
|
+
if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
|
|
1053
|
+
pauseReviewCap(ctx, ex);
|
|
837
1054
|
return;
|
|
838
1055
|
}
|
|
839
|
-
|
|
840
|
-
|
|
1056
|
+
// Phase transition: the executor is done with its tasks; the review loop
|
|
1057
|
+
// owns the run until it converges (or pauses at the cap).
|
|
1058
|
+
setRunStatusForReview(ctx, "verifying");
|
|
1059
|
+
const round: InFlightReview = {
|
|
1060
|
+
controller: new AbortController(),
|
|
1061
|
+
budgetRound: ex.audit.rounds + 1,
|
|
1062
|
+
attempt: ex.review.attempts + 1,
|
|
1063
|
+
fingerprint: captureReviewFingerprint(ctx, ex),
|
|
1064
|
+
wakeSent: false,
|
|
1065
|
+
};
|
|
1066
|
+
ex.review.attempts = round.attempt;
|
|
1067
|
+
ex.review.inFlight = round;
|
|
1068
|
+
ex.audit.running = true; // dashboard mirror (the v0.8 model lands with the overlay task)
|
|
1069
|
+
const spawn = reviewSpawnProfile();
|
|
1070
|
+
reviewModelLabel = spawn.label;
|
|
1071
|
+
const lane = freshReviewLane(round.attempt);
|
|
1072
|
+
reviewLane = lane;
|
|
1073
|
+
reviewOverlay = openReviewOverlay(ctx, lane, ex.uiLanguage, spawn.label);
|
|
841
1074
|
updateStatusWidget(ctx);
|
|
842
|
-
let
|
|
1075
|
+
let result: AuditRoundResult;
|
|
843
1076
|
try {
|
|
844
|
-
|
|
845
|
-
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round:
|
|
1077
|
+
result = auditRunnerForTests
|
|
1078
|
+
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
|
|
846
1079
|
: await runCompletionAudit(ctx, {
|
|
847
1080
|
planPath: ex.planPath,
|
|
848
1081
|
checklist: ex.items,
|
|
849
1082
|
tasks: ex.tasks,
|
|
850
|
-
round:
|
|
851
|
-
|
|
1083
|
+
round: round.attempt,
|
|
1084
|
+
model: spawn.model,
|
|
1085
|
+
thinkingLevel: spawn.thinkingLevel,
|
|
1086
|
+
timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
|
|
1087
|
+
signal: round.controller.signal,
|
|
1088
|
+
onProgress: (event) => {
|
|
1089
|
+
// The engine owns the lane state; the live controller only repaints.
|
|
1090
|
+
applyRefineProgress(lane, event);
|
|
1091
|
+
reviewOverlay?.rerender();
|
|
1092
|
+
},
|
|
1093
|
+
});
|
|
1094
|
+
} catch (error) {
|
|
1095
|
+
if (ex.review.inFlight === round) {
|
|
1096
|
+
ex.review.inFlight = null;
|
|
1097
|
+
ex.audit.running = false;
|
|
1098
|
+
}
|
|
1099
|
+
void reviewOverlay?.close();
|
|
1100
|
+
reviewOverlay = null;
|
|
1101
|
+
messaging().sendMessage(
|
|
1102
|
+
{ customType: "pi-plans-review-error", content: `pi-plans: execution review round threw: ${String(error)}`, display: true },
|
|
1103
|
+
{ triggerTurn: false },
|
|
1104
|
+
);
|
|
1105
|
+
updateStatusWidget(ctx);
|
|
1106
|
+
return;
|
|
1107
|
+
}
|
|
1108
|
+
// Overlay terminal state + close (the controller is one-shot; the engine-held
|
|
1109
|
+
// lane keeps the transcript for a later reopen within this round).
|
|
1110
|
+
const cancelledResult = result !== null && typeof result === "object" && "cancelled" in result;
|
|
1111
|
+
try {
|
|
1112
|
+
applyRefineResult(lane, {
|
|
1113
|
+
ok: !cancelledResult,
|
|
1114
|
+
output: result && "report" in result ? result.report : "",
|
|
1115
|
+
stderr: "",
|
|
1116
|
+
turns: 0,
|
|
1117
|
+
...(cancelledResult ? { cancelled: true as const } : {}),
|
|
1118
|
+
});
|
|
1119
|
+
} catch {
|
|
1120
|
+
/* cosmetic only */
|
|
1121
|
+
}
|
|
1122
|
+
void reviewOverlay?.close();
|
|
1123
|
+
reviewOverlay = null;
|
|
1124
|
+
await handleReviewOutcome(ctx, ex, round, result, pendingChecks.map((item) => item.id));
|
|
1125
|
+
}
|
|
1126
|
+
|
|
1127
|
+
async function handleReviewOutcome(
|
|
1128
|
+
ctx: ExtensionContext,
|
|
1129
|
+
owner: ExecState,
|
|
1130
|
+
round: InFlightReview,
|
|
1131
|
+
result: AuditRoundResult,
|
|
1132
|
+
pendingIds: string[],
|
|
1133
|
+
): Promise<void> {
|
|
1134
|
+
// Identity guard: a restore, stop, or fresh handoff replaced the run —
|
|
1135
|
+
// drop this outcome silently (CF2-002).
|
|
1136
|
+
if (execution !== owner) return;
|
|
1137
|
+
if (owner.review.inFlight !== round) return;
|
|
1138
|
+
owner.review.inFlight = null;
|
|
1139
|
+
owner.audit.running = false;
|
|
1140
|
+
if (result !== null && typeof result === "object" && "cancelled" in result) {
|
|
1141
|
+
// Aborted by shutdown/stop/restore/tree-switch: no round, no budget, no wake.
|
|
1142
|
+
updateStatusWidget(ctx);
|
|
1143
|
+
return;
|
|
1144
|
+
}
|
|
1145
|
+
const outcome = result;
|
|
1146
|
+
const reportText = outcome?.report ?? "(review subagent failed to run)";
|
|
1147
|
+
const fingerprintNow = captureReviewFingerprint(ctx, owner);
|
|
1148
|
+
const coveredTaskIds = flattenTaskViews(owner.tasks).map((task) => task.id);
|
|
1149
|
+
if (fingerprintNow !== round.fingerprint) {
|
|
1150
|
+
// The audited subject moved under the reviewer (Q-C): discard, re-run
|
|
1151
|
+
// without budget burn; two consecutive discards commit as undeterminable
|
|
1152
|
+
// so a mutating user cannot loop the loop for free (Q-discard-bound).
|
|
1153
|
+
owner.review.consecutiveDiscards += 1;
|
|
1154
|
+
const runDir = runDirOf(ctx);
|
|
1155
|
+
if (runDir) {
|
|
1156
|
+
writeReviewRoundReport(runDir, {
|
|
1157
|
+
budgetRound: round.budgetRound,
|
|
1158
|
+
attempt: round.attempt,
|
|
1159
|
+
outcome: "discarded",
|
|
1160
|
+
passed: [],
|
|
1161
|
+
failed: [],
|
|
1162
|
+
undeterminable: pendingIds,
|
|
1163
|
+
discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
|
|
1164
|
+
fingerprintCaptured: round.fingerprint,
|
|
1165
|
+
fingerprintFound: fingerprintNow,
|
|
1166
|
+
coveredTaskIds,
|
|
1167
|
+
report: reportText,
|
|
852
1168
|
});
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
|
|
856
|
-
|
|
857
|
-
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
1169
|
+
}
|
|
1170
|
+
if (owner.review.consecutiveDiscards >= 2) {
|
|
1171
|
+
owner.review.consecutiveDiscards = 0;
|
|
1172
|
+
await commitReviewOutcome(
|
|
1173
|
+
ctx,
|
|
1174
|
+
owner,
|
|
1175
|
+
round,
|
|
1176
|
+
{ passed: [], failed: [], undeterminable: pendingIds, report: `(two consecutive fingerprint discards — committed as an undeterminable round)\n\n${reportText}` },
|
|
1177
|
+
pendingIds,
|
|
1178
|
+
);
|
|
1179
|
+
return;
|
|
1180
|
+
}
|
|
1181
|
+
persist(ctx);
|
|
1182
|
+
updateStatusWidget(ctx);
|
|
1183
|
+
await maybeContinueReview(ctx, owner);
|
|
1184
|
+
return;
|
|
1185
|
+
}
|
|
1186
|
+
owner.review.consecutiveDiscards = 0;
|
|
1187
|
+
await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
|
|
1188
|
+
}
|
|
1189
|
+
|
|
1190
|
+
async function commitReviewOutcome(
|
|
1191
|
+
ctx: ExtensionContext,
|
|
1192
|
+
ex: ExecState,
|
|
1193
|
+
round: InFlightReview,
|
|
1194
|
+
outcome: { passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null,
|
|
1195
|
+
pendingIds: string[],
|
|
1196
|
+
): Promise<void> {
|
|
1197
|
+
// The budget is charged only when an outcome commits — never on discard
|
|
1198
|
+
// or cancellation (CF2-003).
|
|
1199
|
+
ex.audit.rounds = round.budgetRound;
|
|
1200
|
+
const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
|
|
1201
|
+
const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
|
|
1202
|
+
// Anything the round neither passed nor failed is undeterminable: the
|
|
1203
|
+
// report omitted the check, spelled the verdict unreadably, or the
|
|
1204
|
+
// subagent never ran.
|
|
1205
|
+
const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
|
|
1206
|
+
const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
|
|
1207
|
+
const reportText = outcome?.report ?? "(review subagent failed to run)";
|
|
1208
|
+
{
|
|
1209
|
+
const runDir = runDirOf(ctx);
|
|
1210
|
+
if (runDir) {
|
|
1211
|
+
writeReviewRoundReport(runDir, {
|
|
1212
|
+
budgetRound: round.budgetRound,
|
|
1213
|
+
attempt: round.attempt,
|
|
1214
|
+
outcome: outcome === null
|
|
1215
|
+
? "spawn-failed"
|
|
1216
|
+
: failed.length > 0
|
|
1217
|
+
? "failed"
|
|
1218
|
+
: undeterminable.length > 0 ? "undeterminable" : "passed",
|
|
1219
|
+
passed,
|
|
1220
|
+
failed,
|
|
1221
|
+
undeterminable,
|
|
1222
|
+
fingerprintCaptured: round.fingerprint,
|
|
1223
|
+
coveredTaskIds,
|
|
1224
|
+
report: reportText,
|
|
1225
|
+
});
|
|
1226
|
+
}
|
|
1227
|
+
}
|
|
1228
|
+
|
|
1229
|
+
// Fail-closed completion: every pending check affirmatively passed. An
|
|
1230
|
+
// all-undeterminable round yields failed === [] — completing here would be
|
|
1231
|
+
// fail-open, marking a run done with nothing verified.
|
|
1232
|
+
if (passed.length === pendingIds.length) {
|
|
862
1233
|
ex.audit.failed = [];
|
|
1234
|
+
ex.audit.undeterminable = [];
|
|
1235
|
+
for (const id of passed) {
|
|
1236
|
+
const item = ex.items.find((candidate) => candidate.id === id);
|
|
1237
|
+
if (item) item.done = true;
|
|
1238
|
+
}
|
|
863
1239
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
864
1240
|
applyExecutionProgress(cp, {
|
|
865
1241
|
tasks: taskProgressMap(ex.tasks),
|
|
@@ -870,18 +1246,21 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
|
|
|
870
1246
|
await completeExecution(ctx);
|
|
871
1247
|
return;
|
|
872
1248
|
}
|
|
873
|
-
//
|
|
874
|
-
//
|
|
875
|
-
|
|
876
|
-
for (const id of failed) {
|
|
1249
|
+
// Partial progress counts: a check affirmed this round is done even when a
|
|
1250
|
+
// sibling failed, so a later round only re-judges what is still open.
|
|
1251
|
+
for (const id of passed) {
|
|
877
1252
|
const item = ex.items.find((candidate) => candidate.id === id);
|
|
878
|
-
if (item) item.done =
|
|
1253
|
+
if (item) item.done = true;
|
|
879
1254
|
}
|
|
880
1255
|
ex.audit.failed = failed;
|
|
1256
|
+
ex.audit.undeterminable = undeterminable;
|
|
881
1257
|
const rolledBack: string[] = [];
|
|
882
1258
|
for (const id of failed) {
|
|
883
1259
|
rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
|
|
884
1260
|
}
|
|
1261
|
+
// A rollback reopens work other checks were verifying; those checks must
|
|
1262
|
+
// stop claiming the run is satisfied there.
|
|
1263
|
+
if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
|
|
885
1264
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
886
1265
|
applyExecutionProgress(cp, {
|
|
887
1266
|
tasks: taskProgressMap(ex.tasks),
|
|
@@ -889,29 +1268,95 @@ async function runAuditFlow(ctx: ExtensionContext): Promise<void> {
|
|
|
889
1268
|
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") },
|
|
890
1269
|
}),
|
|
891
1270
|
);
|
|
892
|
-
|
|
893
|
-
|
|
1271
|
+
if (rolledBack.length > 0) {
|
|
1272
|
+
// Rolling back is itself forward progress for the watchdog, but NOT for
|
|
1273
|
+
// audit.rounds: that counter stays monotonic so the repair loop is
|
|
1274
|
+
// bounded. The repair belongs to the executor — back to executing.
|
|
1275
|
+
ex.stall.rounds = 0;
|
|
1276
|
+
ex.stall.lastSnapshot = stallSnapshot();
|
|
1277
|
+
setRunStatusForReview(ctx, "executing");
|
|
1278
|
+
}
|
|
894
1279
|
persist(ctx);
|
|
895
1280
|
updateStatusWidget(ctx);
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
1281
|
+
if (failed.length > 0) {
|
|
1282
|
+
// v0.8 wake: per-round one-shot token — exactly one triggerTurn per
|
|
1283
|
+
// committed failed outcome, even when the round resolves long after the
|
|
1284
|
+
// settle that spawned it. The continuation runtime is deliberately
|
|
1285
|
+
// untouched: detached rounds outlive their settle.
|
|
1286
|
+
if (!round.wakeSent) {
|
|
1287
|
+
round.wakeSent = true;
|
|
1288
|
+
const openTasks = flattenTaskViews(ex.tasks)
|
|
1289
|
+
.filter((task) => !taskIsTerminal(task))
|
|
1290
|
+
.map((task) => task.id);
|
|
1291
|
+
const stranded = rolledBack.length === 0;
|
|
1292
|
+
const content = `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}. Rolled back tasks: ${rolledBack.join(", ") || "(none covered)"}. Fix the failures and re-close the rolled-back tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex.audit.rounds >= REVIEW_MAX_ROUNDS ? ` This was round ${REVIEW_MAX_ROUNDS} of ${REVIEW_MAX_ROUNDS}: the next terminal-task cycle pauses the run for review.` : ""}${stranded ? ` No task covers the failed check(s), so the task tree stayed terminal — the next settle re-runs the review automatically.` : ""}`;
|
|
1293
|
+
messaging().sendMessage(
|
|
1294
|
+
{
|
|
1295
|
+
customType: "pi-plans-audit-failed",
|
|
1296
|
+
content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}\n\n---\n${reportText.slice(0, 4000)}`,
|
|
1297
|
+
display: true,
|
|
1298
|
+
},
|
|
1299
|
+
{ triggerTurn: true },
|
|
1300
|
+
);
|
|
1301
|
+
}
|
|
1302
|
+
return; // The agent repairs; the next settle re-enters the loop.
|
|
1303
|
+
}
|
|
1304
|
+
// Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
|
|
1305
|
+
// wake, no message; the dashboard/overlay carries the round counter.
|
|
1306
|
+
await maybeContinueReview(ctx, ex);
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promise<void> {
|
|
1310
|
+
if (execution !== ex) return;
|
|
1311
|
+
if (!reviewOwed(ex)) return;
|
|
1312
|
+
if (ex.stall.paused) return;
|
|
1313
|
+
if (ex.audit.rounds >= REVIEW_MAX_ROUNDS) {
|
|
1314
|
+
pauseReviewCap(ctx, ex);
|
|
1315
|
+
return;
|
|
1316
|
+
}
|
|
1317
|
+
await startReviewRound(ctx);
|
|
1318
|
+
}
|
|
1319
|
+
|
|
1320
|
+
/** Launch a review round under the mode rule: detached (fire-and-forget with
|
|
1321
|
+
* the chain tracked for the test seam) in tui/rpc; awaited inline otherwise
|
|
1322
|
+
* (print/json settle must hold the runtime open through the round). Returns
|
|
1323
|
+
* the chain when the caller must await it, null when detached. */
|
|
1324
|
+
function launchReviewRound(ctx: ExtensionContext): Promise<void> | null {
|
|
1325
|
+
const detach = ctx.mode === "tui" || ctx.mode === "rpc";
|
|
1326
|
+
const chain = (async () => {
|
|
1327
|
+
await startReviewRound(ctx);
|
|
1328
|
+
})();
|
|
1329
|
+
activeReviewChain = chain;
|
|
1330
|
+
if (detach) {
|
|
1331
|
+
chain.catch(() => {
|
|
1332
|
+
/* surfaced via the review messages */
|
|
1333
|
+
});
|
|
1334
|
+
return null;
|
|
1335
|
+
}
|
|
1336
|
+
return chain;
|
|
1337
|
+
}
|
|
1338
|
+
|
|
1339
|
+
/** Deterministic seam: await the in-flight (or self-scheduling) review chain. */
|
|
1340
|
+
export function __awaitReviewRoundForTests(): Promise<void> {
|
|
1341
|
+
return activeReviewChain ?? Promise.resolve();
|
|
1342
|
+
}
|
|
1343
|
+
|
|
1344
|
+
/** When this module graph was first imported into the running pi process.
|
|
1345
|
+
* pi loads extensions once, so a fix written to disk mid-session stays
|
|
1346
|
+
* invisible until /reload — which is exactly why an unreadable audit verdict
|
|
1347
|
+
* deserves a /reload hint rather than a bare retry. */
|
|
1348
|
+
const extensionModuleLoadedAt = new Date();
|
|
1349
|
+
|
|
1350
|
+
/** The /reload advice when the extension on disk is newer than the copy this
|
|
1351
|
+
* process loaded, else null. Shared with /plans via src/staleness.ts so both
|
|
1352
|
+
* report the same answer from one probe. */
|
|
1353
|
+
function staleReloadHint(): string | null {
|
|
1354
|
+
try {
|
|
1355
|
+
const root = path.dirname(path.dirname(new URL(import.meta.url).pathname));
|
|
1356
|
+
return probeStaleReload(root, extensionModuleLoadedAt);
|
|
1357
|
+
} catch {
|
|
1358
|
+
return null;
|
|
1359
|
+
}
|
|
915
1360
|
}
|
|
916
1361
|
|
|
917
1362
|
/** True when the session can surface a pause to a human (D-022): interactive
|
|
@@ -931,7 +1376,7 @@ function activeVccSettings(ctx: ExtensionContext, phase: PiPlansCompactionPhase)
|
|
|
931
1376
|
const run = getRun(ctx.cwd, active.run_id);
|
|
932
1377
|
if (!run) return null;
|
|
933
1378
|
if (phase === "planning" && run.status !== "planning") return null;
|
|
934
|
-
if (phase === "execution" && run.status !== "executing") return null;
|
|
1379
|
+
if (phase === "execution" && run.status !== "executing" && run.status !== "verifying") return null;
|
|
935
1380
|
scaffoldVccSettings(stateRoot);
|
|
936
1381
|
return { settings: loadVccSettings(stateRoot), runId: run.run_id, artifactDir: run.artifact_dir };
|
|
937
1382
|
}
|
|
@@ -1392,6 +1837,9 @@ export function filterPlanningResumeMessages<T extends { customType?: string }>(
|
|
|
1392
1837
|
|
|
1393
1838
|
export async function stopExecution(ctx: ExtensionContext, reason: string): Promise<void> {
|
|
1394
1839
|
if (!execution) return;
|
|
1840
|
+
// A stopped run's in-flight review round dies with it (typed cancelled —
|
|
1841
|
+
// no budget, no wake).
|
|
1842
|
+
abortInFlightReview();
|
|
1395
1843
|
resetExecutionCompactionState(ctx);
|
|
1396
1844
|
pendingExecutionFlush = false;
|
|
1397
1845
|
persist(ctx);
|
|
@@ -1437,7 +1885,7 @@ function stallSnapshot(): string {
|
|
|
1437
1885
|
}
|
|
1438
1886
|
|
|
1439
1887
|
/**
|
|
1440
|
-
* v0.7.1: shared "the
|
|
1888
|
+
* v0.7.1: shared "the execution review is owed" predicate. Every entry point
|
|
1441
1889
|
* that can start the audit (turn_end, agent_before_settle, restoreFromSession,
|
|
1442
1890
|
* the resume path) goes through this so they can never disagree.
|
|
1443
1891
|
*
|
|
@@ -1446,11 +1894,11 @@ function stallSnapshot(): string {
|
|
|
1446
1894
|
*/
|
|
1447
1895
|
function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
|
|
1448
1896
|
if (!ex) return false;
|
|
1449
|
-
// A paused run is never self-driven: the stall /
|
|
1897
|
+
// A paused run is never self-driven: the stall / review-cap pause is an
|
|
1450
1898
|
// explicit "hand control back" signal, and resuming it is the user's call.
|
|
1451
1899
|
// This also bounds the zero-input continue loop in agent_before_settle.
|
|
1452
1900
|
if (ex.stall.paused) return false;
|
|
1453
|
-
if (ex.
|
|
1901
|
+
if (ex.review.inFlight) return false;
|
|
1454
1902
|
return allTasksTerminal(ex.tasks) && auditableChecks(ex.items, ex.tasks).some((item) => !item.done);
|
|
1455
1903
|
}
|
|
1456
1904
|
|
|
@@ -1564,48 +2012,76 @@ export function filterContinuationMessages<T extends { customType?: string; deta
|
|
|
1564
2012
|
return filterGoalWaitMessages(messages);
|
|
1565
2013
|
}
|
|
1566
2014
|
|
|
1567
|
-
/** Prefix of the stall reason used for the audit-cap pause (D-022). */
|
|
1568
|
-
const AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
1569
|
-
|
|
1570
2015
|
/** Called for genuine user input or an explicit same-execution resume.
|
|
1571
|
-
*
|
|
1572
|
-
*
|
|
1573
|
-
*
|
|
2016
|
+
* v0.8: a REVIEW-CAP pause is never lifted here — ordinary input must not
|
|
2017
|
+
* refill the five-round budget (CF2-004); only /plans-execute
|
|
2018
|
+
* (resumeActiveExecution) is the explicit confirmation surface. Genuine
|
|
2019
|
+
* stall pauses still clear on input as before. */
|
|
1574
2020
|
export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
1575
2021
|
const ex = getExecution();
|
|
1576
2022
|
if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
|
|
1577
|
-
|
|
2023
|
+
if (isReviewCapPause(ex.stall.pausedReason)) {
|
|
2024
|
+
// Surfaced once per input so the user is not left guessing why the run
|
|
2025
|
+
// stays paused; the pause itself and the budget survive untouched.
|
|
2026
|
+
ctx.ui.notify?.(
|
|
2027
|
+
"pi-plans: the review-round budget is exhausted — run /plans-execute to grant a fresh five-round budget (that confirmation is the only surface that does).",
|
|
2028
|
+
"warning",
|
|
2029
|
+
);
|
|
2030
|
+
return false;
|
|
2031
|
+
}
|
|
1578
2032
|
ex.stall.paused = false;
|
|
1579
2033
|
ex.stall.pausedReason = undefined;
|
|
1580
2034
|
ex.stall.rounds = 0;
|
|
1581
2035
|
ex.stall.lastSnapshot = stallSnapshot();
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
2036
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null }));
|
|
2037
|
+
persist(ctx);
|
|
2038
|
+
updateStatusWidget(ctx);
|
|
2039
|
+
return true;
|
|
2040
|
+
}
|
|
2041
|
+
|
|
2042
|
+
export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
2043
|
+
// v0.8: /plans-execute is THE explicit confirmation surface for a
|
|
2044
|
+
// review-cap pause — the only place a fresh five-round budget is granted
|
|
2045
|
+
// (Q-confirm-surface). Ordinary input and session restores never refill.
|
|
2046
|
+
const pausedEx = getExecution();
|
|
2047
|
+
if (pausedEx?.stall.paused && isReviewCapPause(pausedEx.stall.pausedReason)) {
|
|
2048
|
+
pausedEx.stall.paused = false;
|
|
2049
|
+
pausedEx.stall.pausedReason = undefined;
|
|
2050
|
+
pausedEx.stall.rounds = 0;
|
|
2051
|
+
pausedEx.stall.lastSnapshot = stallSnapshot();
|
|
2052
|
+
pausedEx.audit.rounds = 0;
|
|
2053
|
+
pausedEx.audit.failed = [];
|
|
2054
|
+
pausedEx.audit.undeterminable = [];
|
|
1585
2055
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
1586
2056
|
applyExecutionProgress(cp, {
|
|
1587
|
-
tasks: taskProgressMap(
|
|
2057
|
+
tasks: taskProgressMap(pausedEx.tasks),
|
|
1588
2058
|
audit: { rounds: 0, lastResult: undefined },
|
|
1589
2059
|
pausedReason: null,
|
|
1590
2060
|
}),
|
|
1591
2061
|
);
|
|
1592
|
-
|
|
1593
|
-
|
|
2062
|
+
persist(ctx);
|
|
2063
|
+
updateStatusWidget(ctx);
|
|
2064
|
+
messaging().sendMessage(
|
|
2065
|
+
{
|
|
2066
|
+
customType: "pi-plans-review-budget-granted",
|
|
2067
|
+
content: "**pi-plans: fresh five-round review budget granted** — the execution review resumes now.",
|
|
2068
|
+
display: true,
|
|
2069
|
+
},
|
|
2070
|
+
{ triggerTurn: false },
|
|
2071
|
+
);
|
|
2072
|
+
const grantChain = launchReviewRound(ctx);
|
|
2073
|
+
if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
|
|
2074
|
+
return true;
|
|
1594
2075
|
}
|
|
1595
|
-
persist(ctx);
|
|
1596
|
-
updateStatusWidget(ctx);
|
|
1597
|
-
return true;
|
|
1598
|
-
}
|
|
1599
|
-
|
|
1600
|
-
export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
1601
2076
|
// v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
|
|
1602
2077
|
// to `return false` here, so `/plans-execute` answered "already executing"
|
|
1603
2078
|
// and the run stayed stranded until a full re-entry or a session restore.
|
|
1604
2079
|
// It is not paused, so the pause path below cannot see it — check it first
|
|
1605
|
-
// and run the owed
|
|
2080
|
+
// and run the owed review instead of reporting "nothing to resume".
|
|
1606
2081
|
if (!execution?.stall.paused && pendingAudit()) {
|
|
1607
2082
|
latchAuditThisSettle();
|
|
1608
|
-
|
|
2083
|
+
const chain = launchReviewRound(ctx);
|
|
2084
|
+
if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
|
|
1609
2085
|
return true;
|
|
1610
2086
|
}
|
|
1611
2087
|
if (!resumeGoalWaitIfPaused(ctx)) return false;
|
|
@@ -1635,7 +2111,7 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
1635
2111
|
messaging().sendMessage(
|
|
1636
2112
|
{
|
|
1637
2113
|
customType: "pi-plans-complete",
|
|
1638
|
-
content: `**Plan complete!** ✅ \`${planPath}\` —
|
|
2114
|
+
content: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}`,
|
|
1639
2115
|
display: true,
|
|
1640
2116
|
},
|
|
1641
2117
|
{ triggerTurn: false },
|
|
@@ -1668,7 +2144,7 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
|
|
|
1668
2144
|
? `${graphBlockForExecutor(false)}\n[pi-plans: config unreadable this turn; graph features are off until .git/pi-plans/config.json is repaired]`
|
|
1669
2145
|
: graphBlockForExecutor(mode === "enabled");
|
|
1670
2146
|
const rollbackNote = execution.audit.failed.length > 0
|
|
1671
|
-
? `\
|
|
2147
|
+
? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
|
|
1672
2148
|
: "";
|
|
1673
2149
|
return `[PI-PLANS EXECUTION — write access enabled]
|
|
1674
2150
|
Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
|
|
@@ -1684,7 +2160,7 @@ Execution rules:
|
|
|
1684
2160
|
- Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
|
|
1685
2161
|
- Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
|
|
1686
2162
|
- Close subtasks before their parent; a parent is auditable only when every child is terminal.
|
|
1687
|
-
- When every task is terminal, the independent
|
|
2163
|
+
- When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
|
|
1688
2164
|
- Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
|
|
1689
2165
|
- Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
|
|
1690
2166
|
- Prefer established, well-maintained libraries when they reduce complexity; check the project's existing dependencies before adding a package or reimplementing common functionality.
|
|
@@ -1745,6 +2221,9 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1745
2221
|
} catch {
|
|
1746
2222
|
tasks = snapshot.tasks;
|
|
1747
2223
|
}
|
|
2224
|
+
// The snapshot cannot carry an in-flight round (it is memory-only); abort
|
|
2225
|
+
// any live one from the previous session graph and rebuild fresh (CF2-002).
|
|
2226
|
+
abortInFlightReview();
|
|
1748
2227
|
execution = {
|
|
1749
2228
|
planPath: snapshot.planPath,
|
|
1750
2229
|
items,
|
|
@@ -1754,8 +2233,14 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1754
2233
|
startedAt: snapshot.startedAt,
|
|
1755
2234
|
usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
|
|
1756
2235
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
1757
|
-
stall: { ...snapshot.stall, lastSnapshot: null
|
|
1758
|
-
audit: {
|
|
2236
|
+
stall: { ...snapshot.stall, lastSnapshot: null },
|
|
2237
|
+
audit: {
|
|
2238
|
+
rounds: snapshot.audit?.rounds ?? 0,
|
|
2239
|
+
failed: snapshot.audit?.failed ?? [],
|
|
2240
|
+
undeterminable: [],
|
|
2241
|
+
running: false,
|
|
2242
|
+
},
|
|
2243
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
1759
2244
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
1760
2245
|
};
|
|
1761
2246
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
@@ -1765,9 +2250,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
1765
2250
|
if (active) bindRun(ctx.sessionManager, ctx.cwd, active.run_id);
|
|
1766
2251
|
persist(ctx);
|
|
1767
2252
|
if (pendingAudit()) {
|
|
1768
|
-
//
|
|
2253
|
+
// Self-heal (v0.8): a verifying run with pending checks restarts its
|
|
2254
|
+
// round exactly once per resume — the full predicate (paused, in-flight,
|
|
2255
|
+
// budget) lives inside startReviewRound. Restores never grant budget.
|
|
1769
2256
|
latchAuditThisSettle();
|
|
1770
|
-
|
|
2257
|
+
const chain = launchReviewRound(ctx);
|
|
2258
|
+
if (chain) await chain;
|
|
1771
2259
|
}
|
|
1772
2260
|
updateStatusWidget(ctx);
|
|
1773
2261
|
}
|