pi-plans 0.8.0 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/agents/execution-reviewer.md +62 -10
- package/package.json +2 -2
- package/references/pi-planning-workflow.md +16 -8
- package/references/state-and-config.md +2 -2
- package/src/auditor.ts +170 -21
- package/src/dashboard.ts +102 -19
- package/src/exec.ts +865 -107
- package/src/refine-ui.ts +21 -5
- package/src/resume-command.ts +26 -6
- package/src/review-budget.ts +290 -0
- package/src/tasks.ts +76 -0
- package/src/ui-language.ts +38 -0
- package/src/workflow-state.ts +217 -8
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +185 -1
- package/tests/dashboard.test.ts +165 -1
- package/tests/exec-review-loop.test.ts +1031 -10
- package/tests/exec.test.ts +32 -5
- package/tests/fixtures/lattice-code-blocked/plan-v2-trimmed.md +57 -0
- package/tests/fixtures/lattice-code-blocked/state.json +158 -0
- package/tests/refine-ui.test.ts +25 -2
- package/tests/resume.test.ts +45 -1
- package/tests/review-budget.test.ts +201 -0
- package/tests/tasks.test.ts +45 -0
- package/tests/workflow-state.test.ts +227 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/execute-plan.ts +12 -5
- package/tools/refine.ts +22 -3
package/src/exec.ts
CHANGED
|
@@ -57,6 +57,7 @@ import {
|
|
|
57
57
|
applyExecutionProgress,
|
|
58
58
|
applyExecutionStopped,
|
|
59
59
|
createCheckpoint,
|
|
60
|
+
applyExecutionPlanAmended,
|
|
60
61
|
loadCheckpoint,
|
|
61
62
|
mutateCheckpoint,
|
|
62
63
|
planIdentityOf,
|
|
@@ -65,6 +66,8 @@ import {
|
|
|
65
66
|
sha256File,
|
|
66
67
|
StaleCheckpointError,
|
|
67
68
|
type ExecutionApproval,
|
|
69
|
+
type ExecutionBlocked,
|
|
70
|
+
type ExecutionCheckpoint,
|
|
68
71
|
type WorkflowCheckpoint,
|
|
69
72
|
} from "./workflow-state.ts";
|
|
70
73
|
import { graphBlockForExecutor } from "./code-graph/prompts.ts";
|
|
@@ -72,6 +75,7 @@ import { resolveGraphMode } from "./code-graph/mode.ts";
|
|
|
72
75
|
import {
|
|
73
76
|
parseChecklist,
|
|
74
77
|
parsePlanTasks,
|
|
78
|
+
CHECKLIST_HEADERS,
|
|
75
79
|
flattenTasks,
|
|
76
80
|
type CheckItem,
|
|
77
81
|
type PlanTasks,
|
|
@@ -80,10 +84,14 @@ import {
|
|
|
80
84
|
allTasksTerminal,
|
|
81
85
|
auditRollbackSet,
|
|
82
86
|
auditableChecks,
|
|
87
|
+
blockedReviewTasks,
|
|
83
88
|
buildTaskView,
|
|
84
89
|
currentTask,
|
|
90
|
+
findingsRollbackSet,
|
|
85
91
|
flattenTaskViews,
|
|
86
92
|
invalidateChecksForRolledBackTasks,
|
|
93
|
+
maxWave,
|
|
94
|
+
rollbackCoverageIds,
|
|
87
95
|
taskIsTerminal,
|
|
88
96
|
taskProgress,
|
|
89
97
|
taskProgressMap,
|
|
@@ -99,13 +107,35 @@ import {
|
|
|
99
107
|
renderDashboardLines,
|
|
100
108
|
renderDashboardTreeLines,
|
|
101
109
|
} from "./dashboard.ts";
|
|
102
|
-
import {
|
|
110
|
+
import { presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
|
|
111
|
+
import {
|
|
112
|
+
DEFAULT_REVIEW_BUDGET,
|
|
113
|
+
LEGACY_REVIEW_MAX_ROUNDS,
|
|
114
|
+
NO_PROGRESS_MAX_STREAK,
|
|
115
|
+
REVIEW_CAP_PAUSE_PREFIX,
|
|
116
|
+
REVIEW_NO_PROGRESS_PAUSE_PREFIX,
|
|
117
|
+
UNLIMITED_HARD_CAP,
|
|
118
|
+
askReviewBudget,
|
|
119
|
+
budgetExhausted,
|
|
120
|
+
bumpNoProgress,
|
|
121
|
+
formatReviewBudget,
|
|
122
|
+
isReviewPauseReason,
|
|
123
|
+
noProgressSignature,
|
|
124
|
+
noProgressTripped,
|
|
125
|
+
resolveStoredBudget,
|
|
126
|
+
reviewBudgetPanelAvailable,
|
|
127
|
+
unlimitedHardCapCeiling,
|
|
128
|
+
type NoProgressState,
|
|
129
|
+
type ReviewBudget,
|
|
130
|
+
} from "./review-budget.ts";
|
|
131
|
+
import type { ReviewFindingRecord } from "./workflow-state.ts";
|
|
103
132
|
import { staleReloadHint as probeStaleReload } from "./staleness.ts";
|
|
104
133
|
import { messaging } from "./messaging.ts";
|
|
105
134
|
import { RefineOverlayController, refineOverlayContext } from "./refine-ui.ts";
|
|
106
135
|
import { applyRefineProgress, applyRefineResult, type RefineLaneState } from "./refine-ui-state.ts";
|
|
107
136
|
import { resolveReviewerSpawn } from "./thinking-levels.ts";
|
|
108
137
|
import { loadGlobalConfig, reviewerReady } from "./global-state.ts";
|
|
138
|
+
import { matchesTerminalKey } from "./terminal-keys.ts";
|
|
109
139
|
|
|
110
140
|
export interface ExecState {
|
|
111
141
|
planPath: string;
|
|
@@ -127,17 +157,52 @@ export interface ExecState {
|
|
|
127
157
|
/** Completion-audit bookkeeping. `rounds` is the BUDGET counter — charged
|
|
128
158
|
* only when a round outcome commits (never on discard/cancel) — and is the
|
|
129
159
|
* only piece persisted. */
|
|
130
|
-
audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean };
|
|
160
|
+
audit: { rounds: number; failed: string[]; undeterminable: string[]; running: boolean; findings: ReviewFinding[] };
|
|
131
161
|
/** Execution-review loop (v0.8), memory-only: the attempt index names the
|
|
132
162
|
* per-round report files; consecutiveDiscards bounds the fingerprint
|
|
133
163
|
* re-run loop; inFlight owns the round's abort lifecycle. */
|
|
134
|
-
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
|
|
164
|
+
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null; budgetAsking?: boolean };
|
|
135
165
|
/** Per-settle audit latch (v0.7.1): a settled round fires the completion
|
|
136
166
|
* audit at most once, so the turn_end / agent_before_settle / resume entry
|
|
137
167
|
* points cannot double-consume a round when several land in one settle.
|
|
138
168
|
* Created on demand by auditLatchOf(); every construction path may omit it. */
|
|
139
169
|
auditLatch?: { auditedThisSettle: boolean; activity: number };
|
|
140
|
-
|
|
170
|
+
/** v0.9.2: outstanding blocker from the newest failed review round — the
|
|
171
|
+
* authoritative rollback set captured at commit time plus the still-open
|
|
172
|
+
* tasks among it. Memory mirror of `execution.blocked`; null = nothing
|
|
173
|
+
* blocks the review. */
|
|
174
|
+
blocked: ExecBlocked | null;
|
|
175
|
+
/** v0.9.2: blocker ids as observed at the PREVIOUS blocked wake — the
|
|
176
|
+
* ladder's progress baseline. Memory-only: `blocked.tasks` is re-synced by
|
|
177
|
+
* `persistTaskProgress` on every task close (so it stays fresh for the
|
|
178
|
+
* resume brief), which would make a comparison against it meaningless.
|
|
179
|
+
* Absent after a restore/resume → the next blocked wake restarts the count. */
|
|
180
|
+
blockedWakeTasks?: string[];
|
|
181
|
+
/** v0.9.2: highest escalation level already surfaced as a visible system
|
|
182
|
+
* line. Memory-only (never persisted/snapshotted) so a restored session
|
|
183
|
+
* re-notifies once. */
|
|
184
|
+
blockedNotifiedLevel?: number;
|
|
185
|
+
/** v0.9.3: the per-run execution-review budget. Resolved exactly once —
|
|
186
|
+
* right before round 1 — from the picker or the no-UI default, then
|
|
187
|
+
* persisted. `undefined` = not decided yet (never conflated with a legacy
|
|
188
|
+
* checkpoint, which `loadExecutionFromCheckpoint` resolves to 5). */
|
|
189
|
+
reviewBudget?: ReviewBudget;
|
|
190
|
+
/** v0.9.3: the budget came from the fallback (no UI/headless/auto-approve),
|
|
191
|
+
* not from a user pick — the expanded dashboard row marks it `(default)`
|
|
192
|
+
* and the resume brief repeats the note. Persisted so a later session can
|
|
193
|
+
* still tell the difference. */
|
|
194
|
+
reviewBudgetDefaulted?: boolean;
|
|
195
|
+
/** v0.9.3: committed review rounds across the whole run (never reset by a
|
|
196
|
+
* grant) — the unlimited budget's cumulative hard-cap counter. */
|
|
197
|
+
reviewRoundsTotal: number;
|
|
198
|
+
/** v0.9.3: rounds added to the unlimited hard cap by explicit grants. */
|
|
199
|
+
reviewCapExtension: number;
|
|
200
|
+
/** v0.9.3: no-progress valve state (unlimited budget only). */
|
|
201
|
+
reviewNoProgress?: NoProgressState;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** Live blocker record (see `ExecutionBlocked` in workflow-state.ts). */
|
|
205
|
+
type ExecBlocked = ExecutionBlocked;
|
|
141
206
|
|
|
142
207
|
/** One in-flight review round: owns its abort lifecycle, its fingerprint of
|
|
143
208
|
* the audited subject, and the per-round one-shot wake token (v0.8). */
|
|
@@ -167,6 +232,8 @@ const STALL_MAX_ROUNDS = 3;
|
|
|
167
232
|
let execution: ExecState | null = null;
|
|
168
233
|
|
|
169
234
|
export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
|
|
235
|
+
/** v0.9.2: visible escalation line — never replays after a restart. */
|
|
236
|
+
export const EXECUTION_BLOCKED_CUSTOM_TYPE = "pi-plans-exec-blocked";
|
|
170
237
|
/** Legacy v0.6.0 continuation message type — filtered on restore. */
|
|
171
238
|
const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
|
|
172
239
|
|
|
@@ -229,9 +296,55 @@ export interface CheckpointExecutionLoad {
|
|
|
229
296
|
/** v0.6.1 (D-020): true when an orphaned v0.6.0 delegated executor was
|
|
230
297
|
* detected — resume requires a fresh handoff approval. */
|
|
231
298
|
legacyDelegate?: boolean;
|
|
299
|
+
/** v0.9.1 (F-005): unresolved findings from the newest committed round,
|
|
300
|
+
* so the /resume-plans brief can surface outstanding highs before the
|
|
301
|
+
* per-turn injection ever runs. */
|
|
302
|
+
findings?: ReviewFinding[];
|
|
303
|
+
/** v0.9.2: persisted blocker of the newest failed round, so the resume
|
|
304
|
+
* brief can name the open tasks that keep the review from starting. */
|
|
305
|
+
blocked?: ExecutionBlocked | null;
|
|
306
|
+
/** v0.9.3: the persisted per-run review budget (or undefined when the run
|
|
307
|
+
* never picked one — a legacy checkpoint resolves to 5 in the live state,
|
|
308
|
+
* see `reviewBudgetDefaulted`). */
|
|
309
|
+
reviewBudget?: ReviewBudget;
|
|
310
|
+
reviewBudgetDefaulted?: boolean;
|
|
311
|
+
reviewRoundsTotal?: number;
|
|
312
|
+
reviewCapExtension?: number;
|
|
232
313
|
error?: string;
|
|
233
314
|
}
|
|
234
315
|
|
|
316
|
+
/** v0.9.2: a checkpoint written before `execution.blocked` existed still names
|
|
317
|
+
* its blocker — `audit.lastResult` records the newest failed round's ids and
|
|
318
|
+
* the coverage cascade is pure, so the rollback set can be reconstructed
|
|
319
|
+
* WITHOUT re-applying anything (the tree already carries the reopen's outcome;
|
|
320
|
+
* re-applying it would revert tasks the executor has since re-closed).
|
|
321
|
+
* Finding-driven rounds (`highs: F-…`) carry no coverage and stay without a
|
|
322
|
+
* record — the wake still states that no round can start while tasks are open. */
|
|
323
|
+
function blockedFromFailedIds(
|
|
324
|
+
failedIds: readonly string[],
|
|
325
|
+
round: number,
|
|
326
|
+
tasks: TaskView[],
|
|
327
|
+
checklist: CheckItem[],
|
|
328
|
+
): ExecBlocked | null {
|
|
329
|
+
if (failedIds.length === 0) return null;
|
|
330
|
+
const rolledBack = rollbackCoverageIds(tasks, checklist, failedIds, []);
|
|
331
|
+
if (rolledBack.length === 0) return null;
|
|
332
|
+
const open = blockedReviewTasks(tasks, rolledBack);
|
|
333
|
+
if (open.length === 0) return null;
|
|
334
|
+
return { rolledBack, tasks: open, round, escalatedRounds: 0, since: utcNow() };
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
function backfillBlocked(
|
|
338
|
+
checkpoint: ExecutionCheckpoint,
|
|
339
|
+
tasks: TaskView[],
|
|
340
|
+
checklist: CheckItem[],
|
|
341
|
+
): ExecBlocked | null {
|
|
342
|
+
const last = checkpoint.audit?.lastResult?.trim() ?? "";
|
|
343
|
+
if (last.length === 0 || last.startsWith("highs:")) return null;
|
|
344
|
+
const failed = last.split(",").map((id) => id.trim()).filter((id) => id.length > 0);
|
|
345
|
+
return blockedFromFailedIds(failed, checkpoint.audit?.rounds ?? 0, tasks, checklist);
|
|
346
|
+
}
|
|
347
|
+
|
|
235
348
|
/**
|
|
236
349
|
* Shared restore primitive: load the executing state from a run checkpoint
|
|
237
350
|
* into THIS session. Authorization is kept only when the recorded approval
|
|
@@ -299,6 +412,11 @@ export function loadExecutionFromCheckpoint(
|
|
|
299
412
|
startedAt: utcNow(),
|
|
300
413
|
usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
|
|
301
414
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
415
|
+
// v0.9.2: the persisted blocker survives a restore so the resume brief
|
|
416
|
+
// and the dashboard can name it; a reverifyAll restore drops it together
|
|
417
|
+
// with the task progress it described. Pre-feature checkpoints are
|
|
418
|
+
// backfilled from the newest failed round's ids.
|
|
419
|
+
blocked: reverifyAll ? null : (cp.execution.blocked ?? backfillBlocked(cp.execution, tasks, items)),
|
|
302
420
|
// D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
|
|
303
421
|
// resume itself is the user's intent; the reason is surfaced in the
|
|
304
422
|
// resume brief instead. EXCEPT a review-cap pause (v0.8): it must
|
|
@@ -314,10 +432,19 @@ export function loadExecutionFromCheckpoint(
|
|
|
314
432
|
rounds: cp.execution.audit?.rounds ?? 0,
|
|
315
433
|
failed: [],
|
|
316
434
|
undeterminable: cp.execution.audit?.undeterminable ?? [],
|
|
435
|
+
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
317
436
|
running: false,
|
|
318
437
|
},
|
|
319
|
-
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
438
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
320
439
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
440
|
+
// v0.9.3: the review budget survives a restore; a checkpoint written
|
|
441
|
+
// before the feature that already spent rounds keeps the legacy 5-round
|
|
442
|
+
// bound instead of being cut to the new default mid-flight.
|
|
443
|
+
reviewBudget: resolveStoredBudget(cp.execution.reviewBudget, cp.execution.audit?.rounds ?? 0),
|
|
444
|
+
reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted === true,
|
|
445
|
+
reviewRoundsTotal: cp.execution.reviewRoundsTotal ?? 0,
|
|
446
|
+
reviewCapExtension: cp.execution.reviewCapExtension ?? 0,
|
|
447
|
+
reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
|
|
321
448
|
};
|
|
322
449
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
323
450
|
executionRunId = runId;
|
|
@@ -352,6 +479,11 @@ export function loadExecutionFromCheckpoint(
|
|
|
352
479
|
if (headChanged) {
|
|
353
480
|
withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
|
|
354
481
|
}
|
|
482
|
+
// A backfilled blocker is authoritative from here on: persist it once so a
|
|
483
|
+
// later restart reads the record instead of re-deriving it.
|
|
484
|
+
if (!reverifyAll && (cp.execution.blocked ?? null) === null && execution!.blocked !== null) {
|
|
485
|
+
withExecutionCheckpoint(ctx, (current) => applyExecutionProgress(current, { blocked: execution!.blocked }));
|
|
486
|
+
}
|
|
355
487
|
persist(ctx);
|
|
356
488
|
updateStatusWidget(ctx);
|
|
357
489
|
return {
|
|
@@ -362,6 +494,14 @@ export function loadExecutionFromCheckpoint(
|
|
|
362
494
|
pausedReason: cp.execution.pausedReason,
|
|
363
495
|
legacyPlan: planTasks.legacy,
|
|
364
496
|
legacyDelegate,
|
|
497
|
+
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
498
|
+
// The LIVE record: a pre-feature checkpoint is backfilled (and persisted)
|
|
499
|
+
// above, so the resume brief sees the reconstructed blocker.
|
|
500
|
+
blocked: execution?.blocked ?? null,
|
|
501
|
+
reviewBudget: execution?.reviewBudget,
|
|
502
|
+
reviewBudgetDefaulted: execution?.reviewBudgetDefaulted === true,
|
|
503
|
+
reviewRoundsTotal: execution?.reviewRoundsTotal ?? 0,
|
|
504
|
+
reviewCapExtension: execution?.reviewCapExtension ?? 0,
|
|
365
505
|
};
|
|
366
506
|
}
|
|
367
507
|
|
|
@@ -501,7 +641,10 @@ function updatePanelWidget(ctx: ExtensionContext): void {
|
|
|
501
641
|
auditRounds: current.audit.rounds > 0 || current.audit.running ? current.audit.rounds : null,
|
|
502
642
|
auditFailed: current.audit.failed,
|
|
503
643
|
auditUndeterminable: current.audit.undeterminable,
|
|
644
|
+
findings: current.audit.findings,
|
|
504
645
|
reviewRunning: current.audit.running === true || current.review.inFlight !== null,
|
|
646
|
+
blockedTasks: blockedTaskIds(current),
|
|
647
|
+
blockedRound: current.blocked?.round ?? null,
|
|
505
648
|
startedAt: current.startedAt,
|
|
506
649
|
usage: current.usage,
|
|
507
650
|
});
|
|
@@ -525,7 +668,14 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
525
668
|
auditRounds: execution.audit.rounds > 0 ? execution.audit.rounds : null,
|
|
526
669
|
auditFailed: execution.audit.failed,
|
|
527
670
|
auditUndeterminable: execution.audit.undeterminable,
|
|
671
|
+
findings: execution.audit.findings,
|
|
528
672
|
reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
|
|
673
|
+
blockedTasks: blockedTaskIds(execution),
|
|
674
|
+
blockedRound: execution.blocked?.round ?? null,
|
|
675
|
+
reviewBudget: execution.reviewBudget ?? null,
|
|
676
|
+
reviewRoundsTotal: execution.reviewRoundsTotal,
|
|
677
|
+
reviewCapExtension: execution.reviewCapExtension,
|
|
678
|
+
reviewBudgetDefaulted: execution.reviewBudgetDefaulted === true,
|
|
529
679
|
});
|
|
530
680
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
|
|
531
681
|
return;
|
|
@@ -597,7 +747,16 @@ function persist(ctx: ExtensionContext): void {
|
|
|
597
747
|
startedAt: execution.startedAt,
|
|
598
748
|
usage: execution.usage,
|
|
599
749
|
stall: execution.stall,
|
|
600
|
-
|
|
750
|
+
blocked: execution.blocked,
|
|
751
|
+
// v0.9.3 (round-1 F-005): the review budget rides the session snapshot
|
|
752
|
+
// too — a /reload restore rebuilds the live state from it, and a
|
|
753
|
+
// restored counter must NOT hand the run a free unlimited window.
|
|
754
|
+
reviewBudget: execution.reviewBudget,
|
|
755
|
+
reviewBudgetDefaulted: execution.reviewBudgetDefaulted,
|
|
756
|
+
reviewRoundsTotal: execution.reviewRoundsTotal,
|
|
757
|
+
reviewCapExtension: execution.reviewCapExtension,
|
|
758
|
+
reviewNoProgress: execution.reviewNoProgress,
|
|
759
|
+
audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
|
|
601
760
|
});
|
|
602
761
|
}
|
|
603
762
|
|
|
@@ -640,9 +799,15 @@ export async function startExecution(
|
|
|
640
799
|
usage: { inToks: 0, outToks: 0 },
|
|
641
800
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
642
801
|
stall: { rounds: 0, lastSnapshot: null, paused: false },
|
|
643
|
-
|
|
644
|
-
|
|
802
|
+
blocked: null,
|
|
803
|
+
audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
|
|
804
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
645
805
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
806
|
+
// v0.9.3: a fresh handoff starts with NO budget — the picker resolves it
|
|
807
|
+
// (or the no-UI default applies) right before round 1.
|
|
808
|
+
reviewBudget: undefined,
|
|
809
|
+
reviewRoundsTotal: 0,
|
|
810
|
+
reviewCapExtension: 0,
|
|
646
811
|
};
|
|
647
812
|
// Seed the watchdog baseline only after `execution` points at the new state
|
|
648
813
|
// (stallSnapshot reads the live execution).
|
|
@@ -702,15 +867,28 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
|
|
|
702
867
|
// Any task-state change resets the stall watchdog baseline.
|
|
703
868
|
execution.stall.rounds = 0;
|
|
704
869
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
870
|
+
// v0.9.2: closing a reopened task shrinks the blocker; when the last one
|
|
871
|
+
// closes the record is dropped so no stale blocker outlives the repair.
|
|
872
|
+
if (execution.blocked) {
|
|
873
|
+
execution.blocked.tasks = blockedReviewTasks(execution.tasks, execution.blocked.rolledBack);
|
|
874
|
+
if (execution.blocked.tasks.length === 0) {
|
|
875
|
+
execution.blocked = null;
|
|
876
|
+
execution.blockedWakeTasks = undefined;
|
|
877
|
+
}
|
|
878
|
+
}
|
|
705
879
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
706
880
|
applyExecutionProgress(cp, {
|
|
707
881
|
tasks: taskProgressMap(execution!.tasks),
|
|
708
882
|
doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
|
|
709
883
|
stallRounds: execution!.stall.rounds,
|
|
884
|
+
blocked: execution!.blocked,
|
|
710
885
|
audit: {
|
|
711
886
|
rounds: execution!.audit.rounds,
|
|
712
887
|
lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
|
|
713
888
|
undeterminable: execution!.audit.undeterminable.length > 0 ? execution!.audit.undeterminable : undefined,
|
|
889
|
+
// v0.9: audit writes are replace-semantics — every writer must
|
|
890
|
+
// carry findings or a task update would silently wipe them.
|
|
891
|
+
findings: execution!.audit.findings,
|
|
714
892
|
},
|
|
715
893
|
}),
|
|
716
894
|
);
|
|
@@ -737,10 +915,10 @@ export function recordExecutionTurn(
|
|
|
737
915
|
}
|
|
738
916
|
|
|
739
917
|
/** Test hook: replace the audit subagent with a deterministic function. */
|
|
740
|
-
let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
|
|
918
|
+
let auditRunnerForTests: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null = null;
|
|
741
919
|
|
|
742
920
|
export function __setAuditRunnerForTests(
|
|
743
|
-
runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
|
|
921
|
+
runner: ((input: { planPath: string; checklist: CheckItem[]; tasks: TaskView[]; round: number; priorFindings: ReviewFinding[] }) => Promise<Awaited<ReturnType<typeof runCompletionAudit>>>) | null,
|
|
744
922
|
): void {
|
|
745
923
|
auditRunnerForTests = runner;
|
|
746
924
|
}
|
|
@@ -867,14 +1045,102 @@ export function registerExecutionTurnHandlers(
|
|
|
867
1045
|
*
|
|
868
1046
|
* Completion stays fail-closed AND never fail-open: a run completes only
|
|
869
1047
|
* when every pending check was affirmatively passed. */
|
|
870
|
-
const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
|
|
871
|
-
/** v0.7 protocol value — dual-matched for one release so checkpoints written
|
|
872
|
-
* by older builds keep their cap pause recognized on restore. */
|
|
873
|
-
const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
874
|
-
|
|
875
1048
|
function isReviewCapPause(reason: string | undefined | null): boolean {
|
|
876
|
-
|
|
877
|
-
|
|
1049
|
+
return isReviewPauseReason(reason);
|
|
1050
|
+
}
|
|
1051
|
+
|
|
1052
|
+
/** v0.9.3: the counters that bound an unlimited budget. */
|
|
1053
|
+
function budgetCounters(ex: ExecState): { reviewRoundsTotal: number; reviewCapExtension: number } {
|
|
1054
|
+
return { reviewRoundsTotal: ex.reviewRoundsTotal, reviewCapExtension: ex.reviewCapExtension };
|
|
1055
|
+
}
|
|
1056
|
+
|
|
1057
|
+
/** The budget in force for this run; a legacy checkpoint resolves to 5 during
|
|
1058
|
+
* load, so this is only a guard for hand-built test states. */
|
|
1059
|
+
function activeBudget(ex: ExecState): ReviewBudget {
|
|
1060
|
+
return ex.reviewBudget ?? LEGACY_REVIEW_MAX_ROUNDS;
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1063
|
+
function budgetSpent(ex: ExecState): boolean {
|
|
1064
|
+
return budgetExhausted(activeBudget(ex), ex.audit.rounds, budgetCounters(ex));
|
|
1065
|
+
}
|
|
1066
|
+
|
|
1067
|
+
/** `3` / `∞` — the budget as shown in status lines and messages. */
|
|
1068
|
+
function budgetLabel(ex: ExecState): string {
|
|
1069
|
+
return formatReviewBudget(activeBudget(ex));
|
|
1070
|
+
}
|
|
1071
|
+
|
|
1072
|
+
/**
|
|
1073
|
+
* v0.9.3: whether the budget panel may be shown in THIS session. Beyond the
|
|
1074
|
+
* native-surface check, auto-approve sessions never ask — the recorded plan
|
|
1075
|
+
* decision routes headless/auto-approve runs to the default budget
|
|
1076
|
+
* (`PI_PLANS_AUTO_APPROVE=1` exists for unattended harnesses, where a panel
|
|
1077
|
+
* would either hang or answer meaninglessly).
|
|
1078
|
+
*/
|
|
1079
|
+
function reviewBudgetPanelUsable(ctx: ExtensionContext): boolean {
|
|
1080
|
+
return !isAutoApproveEnabledLocal() && reviewBudgetPanelAvailable(ctx);
|
|
1081
|
+
}
|
|
1082
|
+
|
|
1083
|
+
/** Persist a freshly resolved budget (and the counters) in one revision. */
|
|
1084
|
+
function persistResolvedBudget(ctx: ExtensionContext, ex: ExecState): void {
|
|
1085
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
1086
|
+
applyExecutionProgress(cp, {
|
|
1087
|
+
reviewBudget: ex.reviewBudget ?? null,
|
|
1088
|
+
reviewBudgetDefaulted: ex.reviewBudgetDefaulted === true,
|
|
1089
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1090
|
+
reviewCapExtension: ex.reviewCapExtension,
|
|
1091
|
+
}),
|
|
1092
|
+
);
|
|
1093
|
+
persist(ctx);
|
|
1094
|
+
updateStatusWidget(ctx);
|
|
1095
|
+
}
|
|
1096
|
+
|
|
1097
|
+
/** One visible note whenever the fallback (rather than a user pick) decided
|
|
1098
|
+
* the budget — never a silent bound (v0.9.3, Q-3). */
|
|
1099
|
+
function notifyDefaultBudget(ex: ExecState, why: "no-panel" | "cancelled"): void {
|
|
1100
|
+
const budget = formatReviewBudget(ex.reviewBudget ?? DEFAULT_REVIEW_BUDGET);
|
|
1101
|
+
messaging().sendMessage(
|
|
1102
|
+
{
|
|
1103
|
+
customType: "pi-plans-review-budget-default",
|
|
1104
|
+
content:
|
|
1105
|
+
why === "no-panel"
|
|
1106
|
+
? `**pi-plans: execution review budget: ${budget} (default)** — no budget panel is available in this session, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`
|
|
1107
|
+
: `**pi-plans: execution review budget: ${budget} (default)** — the budget panel was closed without a choice, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`,
|
|
1108
|
+
display: true,
|
|
1109
|
+
},
|
|
1110
|
+
{ triggerTurn: false },
|
|
1111
|
+
);
|
|
1112
|
+
}
|
|
1113
|
+
|
|
1114
|
+
/** Synchronous fallback: apply (and persist) the default budget. Used when no
|
|
1115
|
+
* panel exists at all — the headless/print/json and auto-approve paths — so
|
|
1116
|
+
* the first round needs no extra async hop. */
|
|
1117
|
+
function applyDefaultReviewBudget(ctx: ExtensionContext, ex: ExecState): void {
|
|
1118
|
+
if (ex.reviewBudget !== undefined) return;
|
|
1119
|
+
ex.reviewBudget = DEFAULT_REVIEW_BUDGET;
|
|
1120
|
+
ex.reviewBudgetDefaulted = true;
|
|
1121
|
+
notifyDefaultBudget(ex, "no-panel");
|
|
1122
|
+
persistResolvedBudget(ctx, ex);
|
|
1123
|
+
}
|
|
1124
|
+
|
|
1125
|
+
/**
|
|
1126
|
+
* Resolve the per-run review budget exactly once, immediately before the first
|
|
1127
|
+
* round. The panel path is async; every other case is handled up front by
|
|
1128
|
+
* `applyDefaultReviewBudget`. `budgetAsking` keeps a second settle (or a
|
|
1129
|
+
* restore while the panel is open) from opening a second panel.
|
|
1130
|
+
*/
|
|
1131
|
+
async function askReviewBudgetForRun(ctx: ExtensionContext, ex: ExecState): Promise<void> {
|
|
1132
|
+
if (ex.reviewBudget !== undefined || ex.review.budgetAsking) return;
|
|
1133
|
+
ex.review.budgetAsking = true;
|
|
1134
|
+
try {
|
|
1135
|
+
const picked = await askReviewBudget(ctx, ex.uiLanguage, undefined);
|
|
1136
|
+
if (execution !== ex || ex.reviewBudget !== undefined) return;
|
|
1137
|
+
ex.reviewBudget = picked ?? DEFAULT_REVIEW_BUDGET;
|
|
1138
|
+
ex.reviewBudgetDefaulted = picked === null;
|
|
1139
|
+
if (picked === null) notifyDefaultBudget(ex, "cancelled");
|
|
1140
|
+
persistResolvedBudget(ctx, ex);
|
|
1141
|
+
} finally {
|
|
1142
|
+
ex.review.budgetAsking = false;
|
|
1143
|
+
}
|
|
878
1144
|
}
|
|
879
1145
|
|
|
880
1146
|
/** The currently-running (or self-scheduling) review chain; the sanctioned
|
|
@@ -894,8 +1160,78 @@ function abortInFlightReview(): void {
|
|
|
894
1160
|
activeReviewChain = null;
|
|
895
1161
|
}
|
|
896
1162
|
|
|
1163
|
+
/** v0.9: unresolved high-severity findings from the newest committed round.
|
|
1164
|
+
* Presence in the newest round's report IS the unresolved set (stable ids:
|
|
1165
|
+
* a problem is resolved only by no longer being reported). */
|
|
1166
|
+
function unresolvedHighFindings(ex: ExecState): ReviewFinding[] {
|
|
1167
|
+
return ex.audit.findings.filter((f) => f.severity === "high");
|
|
1168
|
+
}
|
|
1169
|
+
|
|
1170
|
+
/** v0.9: checkpoint records -> runtime findings. Invalid severities degrade to
|
|
1171
|
+
* "malformed" (recorded, non-blocking) — the same never-fail posture as the
|
|
1172
|
+
* report parser. */
|
|
1173
|
+
function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
|
|
1174
|
+
if (!records) return [];
|
|
1175
|
+
return records.map((r) => {
|
|
1176
|
+
const severity = (r.severity === "high" || r.severity === "medium" || r.severity === "low") ? r.severity : "malformed";
|
|
1177
|
+
return { id: r.id, severity, taskIds: Array.isArray(r.taskIds) ? r.taskIds : [], proposedTask: r.proposedTask, note: r.note ?? "", evidence: r.evidence ?? "", raw: r.raw ?? "" };
|
|
1178
|
+
});
|
|
1179
|
+
}
|
|
1180
|
+
|
|
1181
|
+
/** v0.9: findings widen the owed predicate — an unresolved high finding keeps
|
|
1182
|
+
* the review owed even when every check is done (the stranded-high path),
|
|
1183
|
+
* mirroring how a failed check keeps it owed today. */
|
|
897
1184
|
function reviewOwed(ex: ExecState): boolean {
|
|
898
|
-
return allTasksTerminal(ex.tasks) &&
|
|
1185
|
+
return allTasksTerminal(ex.tasks) && reviewOutstanding(ex);
|
|
1186
|
+
}
|
|
1187
|
+
|
|
1188
|
+
/** v0.9.2: the review is outstanding wherever the tree stands — checks still
|
|
1189
|
+
* owed or an unresolved high finding. `reviewOwed` ANDs this with
|
|
1190
|
+
* allTasksTerminal; the blocked wake needs exactly the half that stays true
|
|
1191
|
+
* while tasks are open, because that is the state in which the review cannot
|
|
1192
|
+
* start (and in which `pendingAudit` is false by construction). */
|
|
1193
|
+
function reviewOutstanding(ex: ExecState): boolean {
|
|
1194
|
+
return auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0;
|
|
1195
|
+
}
|
|
1196
|
+
|
|
1197
|
+
/** v0.9.2: live blocker ids — the newest failed round's rollback set
|
|
1198
|
+
* intersected with the still-open tasks. Always recomputed from the tree: the
|
|
1199
|
+
* stored `tasks` list may predate the last status change. */
|
|
1200
|
+
function blockedTaskIds(ex: ExecState): string[] {
|
|
1201
|
+
return ex.blocked ? blockedReviewTasks(ex.tasks, ex.blocked.rolledBack) : [];
|
|
1202
|
+
}
|
|
1203
|
+
|
|
1204
|
+
/** v0.9.2: one visible system line per escalation level, so the user sees the
|
|
1205
|
+
* loop escalating before the watchdog pauses it. Memory-only latch (the
|
|
1206
|
+
* restored state re-notifies once, which is the useful behavior). */
|
|
1207
|
+
function notifyBlockedEscalation(ctx: ExtensionContext, ex: ExecState): void {
|
|
1208
|
+
const level = ex.blocked?.escalatedRounds ?? 0;
|
|
1209
|
+
if (!ex.blocked || level === 0 || ex.blockedNotifiedLevel === level) return;
|
|
1210
|
+
ex.blockedNotifiedLevel = level;
|
|
1211
|
+
const ids = blockedTaskIds(ex);
|
|
1212
|
+
try {
|
|
1213
|
+
messaging().sendMessage(
|
|
1214
|
+
{
|
|
1215
|
+
customType: EXECUTION_BLOCKED_CUSTOM_TYPE,
|
|
1216
|
+
content: `**pi-plans: execution blocked (wake ${level}/${STALL_MAX_ROUNDS})** — review round ${ex.blocked.round + 1} cannot start: ${ids.join(", ")} ${ids.length === 1 ? "was" : "were"} reopened by round ${ex.blocked.round} and ${ids.length === 1 ? "is" : "are"} still open. Close them with \`plans_update_task\` (the review starts by itself once every task is terminal).`,
|
|
1217
|
+
display: true,
|
|
1218
|
+
},
|
|
1219
|
+
{ triggerTurn: false },
|
|
1220
|
+
);
|
|
1221
|
+
} catch {
|
|
1222
|
+
/* best-effort: the wake below still carries the same blocker */
|
|
1223
|
+
}
|
|
1224
|
+
}
|
|
1225
|
+
|
|
1226
|
+
/** v0.9.2: the pause reason names the real blocker instead of the watchdog's
|
|
1227
|
+
* own metric. Single line, never a review-cap prefix (`isReviewCapPause`
|
|
1228
|
+
* matches "execution review exhausted" / "completion audit exhausted"). */
|
|
1229
|
+
function blockedPauseReason(ex: ExecState): string {
|
|
1230
|
+
const ids = blockedTaskIds(ex);
|
|
1231
|
+
const round = ex.blocked?.round ?? ex.audit.rounds;
|
|
1232
|
+
const head = ids.slice(0, 3).join(", ");
|
|
1233
|
+
const more = ids.length > 3 ? `, +${ids.length - 3} more` : "";
|
|
1234
|
+
return `blocked: review round ${round + 1} cannot start — ${ids.length} task(s) reopened by round ${round} still open (${head}${more})`;
|
|
899
1235
|
}
|
|
900
1236
|
|
|
901
1237
|
function runDirOf(ctx: ExtensionContext): string | null {
|
|
@@ -996,11 +1332,21 @@ function freshReviewLane(attempt: number): RefineLaneState {
|
|
|
996
1332
|
|
|
997
1333
|
/** Fresh controller per round (the controller is one-shot: closed latch,
|
|
998
1334
|
* overlayPromise bail, terminal-lane early return — reuse drops progress).
|
|
999
|
-
* A UI failure must NEVER kill the round itself — best-effort only.
|
|
1335
|
+
* A UI failure must NEVER kill the round itself — best-effort only. The
|
|
1336
|
+
* overlay forwards unhandled keys so Ctrl+Shift+T keeps working while the
|
|
1337
|
+
* review overlay holds focus (pi-tui has no key bubbling). */
|
|
1000
1338
|
function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "en" | "zh" | undefined, modelLabel: string): RefineOverlayController | null {
|
|
1001
1339
|
if (ctx.mode !== "tui" || ctx.hasUI !== true) return null;
|
|
1002
1340
|
try {
|
|
1003
|
-
const controller = new RefineOverlayController(
|
|
1341
|
+
const controller = new RefineOverlayController(
|
|
1342
|
+
"auditor",
|
|
1343
|
+
[{ id: lane.id, label: lane.label }],
|
|
1344
|
+
() => {},
|
|
1345
|
+
lang ?? "en",
|
|
1346
|
+
(data) => {
|
|
1347
|
+
if (matchesTerminalKey(data, "ctrl+shift+t")) toggleDashboardExpanded(ctx);
|
|
1348
|
+
},
|
|
1349
|
+
);
|
|
1004
1350
|
controller.seedLane(lane);
|
|
1005
1351
|
controller.open(refineOverlayContext(ctx), modelLabel);
|
|
1006
1352
|
return controller;
|
|
@@ -1013,16 +1359,27 @@ function openReviewOverlay(ctx: ExtensionContext, lane: RefineLaneState, lang: "
|
|
|
1013
1359
|
* state; inert when no round is in flight. */
|
|
1014
1360
|
export function reopenReviewOverlay(ctx: ExtensionContext): void {
|
|
1015
1361
|
if (!execution?.review.inFlight || !reviewLane) return;
|
|
1362
|
+
// Never stack a second overlay on a live one (Ctrl+Shift+R while open).
|
|
1363
|
+
if (reviewOverlay && !reviewOverlay.isClosed()) return;
|
|
1016
1364
|
const controller = openReviewOverlay(ctx, reviewLane, execution.uiLanguage, reviewModelLabel ?? "session default");
|
|
1017
1365
|
if (controller) reviewOverlay = controller;
|
|
1018
1366
|
}
|
|
1019
1367
|
|
|
1020
1368
|
function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
1369
|
+
const highs = unresolvedHighFindings(ex);
|
|
1021
1370
|
const detail =
|
|
1022
1371
|
ex.audit.failed.length > 0
|
|
1023
|
-
? `failed: ${ex.audit.failed.join(", ")}`
|
|
1024
|
-
:
|
|
1025
|
-
|
|
1372
|
+
? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
|
|
1373
|
+
: highs.length > 0
|
|
1374
|
+
? `high findings: ${highs.map((f) => f.id).join(", ")}`
|
|
1375
|
+
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1376
|
+
// v0.9.3: the reason names the budget actually in force — the numeric round
|
|
1377
|
+
// count, or the unlimited hard cap that stopped the loop.
|
|
1378
|
+
const scope =
|
|
1379
|
+
activeBudget(ex) === "unlimited"
|
|
1380
|
+
? `${unlimitedHardCapCeiling(budgetCounters(ex))} rounds (unlimited budget safety cap)`
|
|
1381
|
+
: `${activeBudget(ex)} round${activeBudget(ex) === 1 ? "" : "s"}`;
|
|
1382
|
+
const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${scope} (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the ${formatReviewBudget(activeBudget(ex))} picker; ordinary messages and session restores do not). (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
|
|
1026
1383
|
pauseForStall(ctx, reason);
|
|
1027
1384
|
// In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
|
|
1028
1385
|
// pauseForStall's ui.notify is optional and absent headless, so the pause
|
|
@@ -1033,6 +1390,41 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
|
1033
1390
|
);
|
|
1034
1391
|
}
|
|
1035
1392
|
|
|
1393
|
+
/** v0.9.3: the unlimited budget's no-progress valve — three consecutive
|
|
1394
|
+
* committed rounds with an identical outcome signature cannot converge, so
|
|
1395
|
+
* the run pauses fail-closed instead of burning rounds silently. */
|
|
1396
|
+
function pauseReviewNoProgress(ctx: ExtensionContext, ex: ExecState): void {
|
|
1397
|
+
const streak = ex.reviewNoProgress?.streak ?? NO_PROGRESS_MAX_STREAK;
|
|
1398
|
+
const highs = unresolvedHighFindings(ex);
|
|
1399
|
+
const detail =
|
|
1400
|
+
ex.audit.failed.length > 0
|
|
1401
|
+
? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
|
|
1402
|
+
: highs.length > 0
|
|
1403
|
+
? `high findings: ${highs.map((f) => f.id).join(", ")}`
|
|
1404
|
+
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1405
|
+
const reason = `${REVIEW_NO_PROGRESS_PAUSE_PREFIX} — ${streak} consecutive rounds reported the same outcome (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the budget picker; ordinary messages and session restores do not).`;
|
|
1406
|
+
pauseForStall(ctx, reason);
|
|
1407
|
+
messaging().sendMessage(
|
|
1408
|
+
{ customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
|
|
1409
|
+
{ triggerTurn: false },
|
|
1410
|
+
);
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
/** v0.9.3: the single gate every round spawn passes through. Returns false
|
|
1414
|
+
* when the run was paused (no round may start). The unlimited valve order is
|
|
1415
|
+
* deliberate: no-progress is checked first so its reason wins when both hold. */
|
|
1416
|
+
function reviewBudgetGate(ctx: ExtensionContext, ex: ExecState): boolean {
|
|
1417
|
+
if (activeBudget(ex) === "unlimited" && noProgressTripped(ex.reviewNoProgress)) {
|
|
1418
|
+
pauseReviewNoProgress(ctx, ex);
|
|
1419
|
+
return false;
|
|
1420
|
+
}
|
|
1421
|
+
if (budgetSpent(ex)) {
|
|
1422
|
+
pauseReviewCap(ctx, ex);
|
|
1423
|
+
return false;
|
|
1424
|
+
}
|
|
1425
|
+
return true;
|
|
1426
|
+
}
|
|
1427
|
+
|
|
1036
1428
|
async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
1037
1429
|
if (!execution) return;
|
|
1038
1430
|
const ex = execution;
|
|
@@ -1044,15 +1436,34 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
|
1044
1436
|
// Only auditable checks (with task coverage) gate completion; checks that
|
|
1045
1437
|
// cover no task can never be verified and never block or complete.
|
|
1046
1438
|
const pendingChecks = auditableChecks(ex.items, ex.tasks).filter((item) => !item.done);
|
|
1047
|
-
|
|
1439
|
+
// v0.9: unresolved high findings block the fast completion path — they
|
|
1440
|
+
// keep the review owed instead (liveness: a high can never be completed
|
|
1441
|
+
// around, only fixed or paused at the cap).
|
|
1442
|
+
if (pendingChecks.length === 0 && unresolvedHighFindings(ex).length === 0) {
|
|
1048
1443
|
await completeExecution(ctx);
|
|
1049
1444
|
return;
|
|
1050
1445
|
}
|
|
1051
|
-
if (ex.stall.paused || ex.review.inFlight) return;
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1446
|
+
if (ex.stall.paused || ex.review.inFlight || ex.review.budgetAsking) return;
|
|
1447
|
+
// v0.9.3: the budget is resolved right here — every task is terminal, the
|
|
1448
|
+
// review is genuinely owed, and this is the last moment before round 1.
|
|
1449
|
+
// No panel → the default applies synchronously (no extra tick, so a
|
|
1450
|
+
// detached settle still shows the round in flight immediately); a panel →
|
|
1451
|
+
// one await, guarded against a second settle by `budgetAsking`.
|
|
1452
|
+
if (ex.reviewBudget === undefined && !reviewBudgetPanelUsable(ctx)) applyDefaultReviewBudget(ctx, ex);
|
|
1453
|
+
if (ex.reviewBudget === undefined) {
|
|
1454
|
+
await askReviewBudgetForRun(ctx, ex);
|
|
1455
|
+
if (execution !== ex || ex.reviewBudget === undefined) return;
|
|
1456
|
+
}
|
|
1457
|
+
if (pendingChecks.length === 0) {
|
|
1458
|
+
// Every verification check is satisfied; only unresolved highs keep the
|
|
1459
|
+
// review owed. With the budget spent there is no round left to buy, so
|
|
1460
|
+
// the run completes and DISCLOSES the highs (v0.9.3, Q-2).
|
|
1461
|
+
if (budgetSpent(ex)) {
|
|
1462
|
+
await completeExecution(ctx);
|
|
1463
|
+
return;
|
|
1464
|
+
}
|
|
1055
1465
|
}
|
|
1466
|
+
if (!reviewBudgetGate(ctx, ex)) return;
|
|
1056
1467
|
// Phase transition: the executor is done with its tasks; the review loop
|
|
1057
1468
|
// owns the run until it converges (or pauses at the cap).
|
|
1058
1469
|
setRunStatusForReview(ctx, "verifying");
|
|
@@ -1075,12 +1486,13 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
|
1075
1486
|
let result: AuditRoundResult;
|
|
1076
1487
|
try {
|
|
1077
1488
|
result = auditRunnerForTests
|
|
1078
|
-
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt })
|
|
1489
|
+
? await auditRunnerForTests({ planPath: ex.planPath, checklist: ex.items, tasks: ex.tasks, round: round.attempt, priorFindings: ex.audit.findings })
|
|
1079
1490
|
: await runCompletionAudit(ctx, {
|
|
1080
1491
|
planPath: ex.planPath,
|
|
1081
1492
|
checklist: ex.items,
|
|
1082
1493
|
tasks: ex.tasks,
|
|
1083
1494
|
round: round.attempt,
|
|
1495
|
+
priorFindings: ex.audit.findings,
|
|
1084
1496
|
model: spawn.model,
|
|
1085
1497
|
thinkingLevel: spawn.thinkingLevel,
|
|
1086
1498
|
timeoutMs: REVIEW_ROUND_TIMEOUT_MS,
|
|
@@ -1160,6 +1572,9 @@ async function handleReviewOutcome(
|
|
|
1160
1572
|
passed: [],
|
|
1161
1573
|
failed: [],
|
|
1162
1574
|
undeterminable: pendingIds,
|
|
1575
|
+
// v0.9.1 (F-014): the discard report must not understate a
|
|
1576
|
+
// round whose embedded report carries highs.
|
|
1577
|
+
findings: owner.audit.findings,
|
|
1163
1578
|
discardedReason: "fingerprint changed between round start and resolve (worktree/plan moved under the reviewer)",
|
|
1164
1579
|
fingerprintCaptured: round.fingerprint,
|
|
1165
1580
|
fingerprintFound: fingerprintNow,
|
|
@@ -1187,113 +1602,265 @@ async function handleReviewOutcome(
|
|
|
1187
1602
|
await commitReviewOutcome(ctx, owner, round, outcome, pendingIds);
|
|
1188
1603
|
}
|
|
1189
1604
|
|
|
1605
|
+
/** v0.9 (findings-driven fix loop): append plan tasks for high findings no
|
|
1606
|
+
* existing task owns. The reviewer stays read-only — it proposes the title
|
|
1607
|
+
* (proposed-task); this machinery applies it with provenance, so every high
|
|
1608
|
+
* finding always has an owner the executor can close, and the stable F-###
|
|
1609
|
+
* id rides in the task title for traceability. Best-effort: an unwritable
|
|
1610
|
+
* plan must not crash the loop (the finding then stays stranded and the cap
|
|
1611
|
+
* pause surfaces it). Returns the appended task ids. */
|
|
1612
|
+
function appendFindingTasks(ex: ExecState, highs: ReviewFinding[]): string[] {
|
|
1613
|
+
const appended: string[] = [];
|
|
1614
|
+
let next = flattenTaskViews(ex.tasks).reduce((max, t) => {
|
|
1615
|
+
const m = /^Task-(\d+)$/.exec(t.id);
|
|
1616
|
+
return m ? Math.max(max, Number(m[1])) : max;
|
|
1617
|
+
}, 0);
|
|
1618
|
+
const wave = maxWave(ex.tasks) + 1;
|
|
1619
|
+
try {
|
|
1620
|
+
const planText = fs.readFileSync(ex.planPath, "utf8");
|
|
1621
|
+
const lines = planText.split("\n");
|
|
1622
|
+
// Insert inside the Tasks section: before the Execution Waves
|
|
1623
|
+
// subsection when present, else before the FIRST checklist header the
|
|
1624
|
+
// plan uses (v0.9.1 F-013: legacy plans say `## Verifier Checklist`,
|
|
1625
|
+
// and an EOF fallback would land outside every parsed section), else
|
|
1626
|
+
// at EOF; walk back over blank separators so the bullet lands
|
|
1627
|
+
// adjacent to its siblings.
|
|
1628
|
+
let insertAt = lines.length;
|
|
1629
|
+
const wavesIdx = lines.findIndex((l) => /^###\s+Execution Waves/.test(l));
|
|
1630
|
+
const checklistIdx = lines.findIndex((l) => CHECKLIST_HEADERS.some((h) => new RegExp(`^##\\s+${h}`).test(l)));
|
|
1631
|
+
if (wavesIdx !== -1) insertAt = wavesIdx;
|
|
1632
|
+
else if (checklistIdx !== -1) insertAt = checklistIdx;
|
|
1633
|
+
while (insertAt > 0 && lines[insertAt - 1].trim() === "") insertAt--;
|
|
1634
|
+
// v0.9.1 (F-012): reviewer text becomes task-title metadata at parse
|
|
1635
|
+
// time — strip the microsyntax metacharacters (em/en dashes, `--`
|
|
1636
|
+
// separators, the `;` field delimiter) so an embedded token can never
|
|
1637
|
+
// split title from tail or forge fields.
|
|
1638
|
+
const sanitize = (text: string): string => text.replace(/[—–]/g, "-").replace(/-{2,}/g, "-").replace(/;/g, ",");
|
|
1639
|
+
const entries: Array<{ id: string; title: string }> = [];
|
|
1640
|
+
const newLines: string[] = [];
|
|
1641
|
+
for (const h of highs) {
|
|
1642
|
+
next += 1;
|
|
1643
|
+
const id = `Task-${next}`;
|
|
1644
|
+
const title = `fix ${h.id}: ${sanitize(h.proposedTask ?? h.note ?? "address the finding")} (appended by execution review round ${ex.audit.rounds})`;
|
|
1645
|
+
// v0.9.1 (F-006): carry the wave in the bullet tail so a re-parse
|
|
1646
|
+
// restores the same wave the live tree assigned — without it the
|
|
1647
|
+
// appended remediation task fell back to wave 1 on restore and
|
|
1648
|
+
// hijacked the ▸ anchor.
|
|
1649
|
+
newLines.push(`- \`${id}\`: ${title} — wave: ${wave}`);
|
|
1650
|
+
entries.push({ id, title });
|
|
1651
|
+
}
|
|
1652
|
+
lines.splice(insertAt, 0, ...newLines);
|
|
1653
|
+
fs.writeFileSync(ex.planPath, lines.join("\n"), "utf8");
|
|
1654
|
+
// v0.9.1 (F-008): only a successful plan write mints the live tasks —
|
|
1655
|
+
// pushing before the write left checkpoint entries the plan file does
|
|
1656
|
+
// not contain whenever the write failed.
|
|
1657
|
+
for (const entry of entries) {
|
|
1658
|
+
ex.tasks.push({ id: entry.id, title: entry.title, wave, deps: [], files: [], status: "pending", children: [] });
|
|
1659
|
+
appended.push(entry.id);
|
|
1660
|
+
}
|
|
1661
|
+
} catch {
|
|
1662
|
+
/* best-effort: stranded highs surface via the cap pause */
|
|
1663
|
+
}
|
|
1664
|
+
return appended;
|
|
1665
|
+
}
|
|
1666
|
+
|
|
1190
1667
|
async function commitReviewOutcome(
|
|
1191
1668
|
ctx: ExtensionContext,
|
|
1192
1669
|
ex: ExecState,
|
|
1193
1670
|
round: InFlightReview,
|
|
1194
|
-
outcome:
|
|
1671
|
+
outcome: AuditOutcome | null,
|
|
1195
1672
|
pendingIds: string[],
|
|
1196
1673
|
): Promise<void> {
|
|
1197
1674
|
// The budget is charged only when an outcome commits — never on discard
|
|
1198
1675
|
// or cancellation (CF2-003).
|
|
1199
1676
|
ex.audit.rounds = round.budgetRound;
|
|
1677
|
+
// v0.9.3: the run-cumulative counter never resets — it is what bounds an
|
|
1678
|
+
// unlimited budget across grants (Q-4).
|
|
1679
|
+
ex.reviewRoundsTotal += 1;
|
|
1200
1680
|
const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
|
|
1201
1681
|
const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
|
|
1202
1682
|
// Anything the round neither passed nor failed is undeterminable: the
|
|
1203
1683
|
// report omitted the check, spelled the verdict unreadably, or the
|
|
1204
1684
|
// subagent never ran.
|
|
1205
1685
|
const undeterminable = pendingIds.filter((id) => !passed.includes(id) && !failed.includes(id));
|
|
1686
|
+
// v0.9: the newest round's reported findings ARE the unresolved set
|
|
1687
|
+
// (stable ids — a problem is resolved only by no longer being reported).
|
|
1688
|
+
// v0.9.1 (F-001): only a REAL parsed report is authoritative. A round that
|
|
1689
|
+
// produced no report at all — spawn failure (outcome === null) or the
|
|
1690
|
+
// two-consecutive-discard synthesis — must PRESERVE the unresolved set:
|
|
1691
|
+
// clearing it let the vacuous completion guard (empty pendingIds) mark a
|
|
1692
|
+
// run done with its high finding silently dropped.
|
|
1693
|
+
const reported = outcome !== null && outcome.findings !== undefined;
|
|
1694
|
+
const findings = reported ? outcome.findings : ex.audit.findings;
|
|
1695
|
+
ex.audit.findings = findings;
|
|
1696
|
+
const highs = findings.filter((f) => f.severity === "high");
|
|
1697
|
+
// v0.9.3: the no-progress valve's signature is computed HERE, from the
|
|
1698
|
+
// post-classification triple — a spawn-failure round (`outcome === null`)
|
|
1699
|
+
// and a discard synthesis have no findings array of their own, but the
|
|
1700
|
+
// preserved unresolved set plus the derived verdicts still describe a
|
|
1701
|
+
// concrete, comparable outcome (round-1 F-002).
|
|
1702
|
+
ex.reviewNoProgress = bumpNoProgress(ex.reviewNoProgress, noProgressSignature(failed, undeterminable, highs.map((f) => f.id)));
|
|
1703
|
+
// Findings are actionable only when this round actually reported them;
|
|
1704
|
+
// see the fix-loop branch below (v0.9.1, F-001).
|
|
1705
|
+
const actionableHighs = reported ? highs : [];
|
|
1206
1706
|
const coveredTaskIds = flattenTaskViews(ex.tasks).map((task) => task.id);
|
|
1207
1707
|
const reportText = outcome?.report ?? "(review subagent failed to run)";
|
|
1708
|
+
let reportPath: string | null = null;
|
|
1208
1709
|
{
|
|
1209
1710
|
const runDir = runDirOf(ctx);
|
|
1210
1711
|
if (runDir) {
|
|
1211
|
-
writeReviewRoundReport(runDir, {
|
|
1712
|
+
reportPath = writeReviewRoundReport(runDir, {
|
|
1212
1713
|
budgetRound: round.budgetRound,
|
|
1213
1714
|
attempt: round.attempt,
|
|
1214
1715
|
outcome: outcome === null
|
|
1215
1716
|
? "spawn-failed"
|
|
1216
|
-
: failed.length > 0
|
|
1717
|
+
: failed.length > 0 || actionableHighs.length > 0
|
|
1217
1718
|
? "failed"
|
|
1218
1719
|
: undeterminable.length > 0 ? "undeterminable" : "passed",
|
|
1219
1720
|
passed,
|
|
1220
1721
|
failed,
|
|
1221
1722
|
undeterminable,
|
|
1723
|
+
findings,
|
|
1222
1724
|
fingerprintCaptured: round.fingerprint,
|
|
1223
1725
|
coveredTaskIds,
|
|
1224
1726
|
report: reportText,
|
|
1225
1727
|
});
|
|
1226
1728
|
}
|
|
1227
1729
|
}
|
|
1730
|
+
// Partial progress counts: a check affirmed this round is done even when
|
|
1731
|
+
// a sibling failed, so a later round only re-judges what is still open.
|
|
1732
|
+
for (const id of passed) {
|
|
1733
|
+
const item = ex.items.find((candidate) => candidate.id === id);
|
|
1734
|
+
if (item) item.done = true;
|
|
1735
|
+
}
|
|
1228
1736
|
|
|
1229
|
-
//
|
|
1230
|
-
//
|
|
1231
|
-
//
|
|
1232
|
-
|
|
1737
|
+
// v0.9.3 (Q-2/F-003): the exhausted-budget completion. EVERY check this
|
|
1738
|
+
// round still owed affirmed, only unresolved highs remain, and the budget
|
|
1739
|
+
// cannot buy another round — the run completes, tolerating the highs (they
|
|
1740
|
+
// stay in the round report and the checkpoint; `completeExecution`
|
|
1741
|
+
// discloses them). Evaluated BEFORE the fix branch on purpose: the
|
|
1742
|
+
// tolerated round must not roll back tasks, append plan tasks, rewrite the
|
|
1743
|
+
// approved plan, or wake the executor for a run that is about to be done.
|
|
1744
|
+
if (budgetSpent(ex) && passed.length === pendingIds.length && highs.length > 0) {
|
|
1233
1745
|
ex.audit.failed = [];
|
|
1234
1746
|
ex.audit.undeterminable = [];
|
|
1235
|
-
|
|
1236
|
-
const item = ex.items.find((candidate) => candidate.id === id);
|
|
1237
|
-
if (item) item.done = true;
|
|
1238
|
-
}
|
|
1747
|
+
ex.blocked = null;
|
|
1239
1748
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
1240
1749
|
applyExecutionProgress(cp, {
|
|
1241
1750
|
tasks: taskProgressMap(ex.tasks),
|
|
1242
1751
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1243
|
-
|
|
1752
|
+
blocked: null,
|
|
1753
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1754
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1755
|
+
audit: { rounds: ex.audit.rounds, passed: true, findings },
|
|
1244
1756
|
}),
|
|
1245
1757
|
);
|
|
1758
|
+
persist(ctx);
|
|
1759
|
+
messaging().sendMessage(
|
|
1760
|
+
{
|
|
1761
|
+
customType: "pi-plans-review-tolerated",
|
|
1762
|
+
content: `**pi-plans: review budget exhausted (${budgetLabel(ex)}) — completing with ${highs.length} unresolved high finding(s): ${highs.map((f) => f.id).join(", ")}** — every verification check passed; the finding(s) remain in the round reports.`,
|
|
1763
|
+
display: true,
|
|
1764
|
+
},
|
|
1765
|
+
{ triggerTurn: false },
|
|
1766
|
+
);
|
|
1246
1767
|
await completeExecution(ctx);
|
|
1247
1768
|
return;
|
|
1248
1769
|
}
|
|
1249
|
-
|
|
1250
|
-
//
|
|
1251
|
-
|
|
1252
|
-
|
|
1253
|
-
|
|
1254
|
-
|
|
1255
|
-
|
|
1256
|
-
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1260
|
-
|
|
1261
|
-
|
|
1262
|
-
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
|
|
1285
|
-
|
|
1770
|
+
|
|
1771
|
+
// v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
|
|
1772
|
+
// high finding can never complete the run (liveness). Failed checks and
|
|
1773
|
+
// high findings drive ONE union rollback and exactly one executor wake;
|
|
1774
|
+
// high wins over the undeterminable self-schedule (a finding is
|
|
1775
|
+
// actionable independent of verdict evidence). v0.9.1 (F-001): verdicts
|
|
1776
|
+
// are authoritative whenever a report exists, but the FINDINGS-driven
|
|
1777
|
+
// half of the branch needs a findings-bearing report — a no-findings
|
|
1778
|
+
// round (spawn failure, discard synthesis, legacy shape) preserves the
|
|
1779
|
+
// unresolved set and self-schedules instead of rolling back on it.
|
|
1780
|
+
if (failed.length > 0 || actionableHighs.length > 0) {
|
|
1781
|
+
ex.audit.failed = failed;
|
|
1782
|
+
ex.audit.undeterminable = undeterminable;
|
|
1783
|
+
const rolledBack: string[] = [];
|
|
1784
|
+
for (const id of failed) {
|
|
1785
|
+
rolledBack.push(...auditRollbackSet(ex.tasks, ex.items, id));
|
|
1786
|
+
}
|
|
1787
|
+
// Invalidation is the VC-fail invariant ONLY: a check re-verifies its
|
|
1788
|
+
// reopened tasks when a FAILED check rolled them back. A pure
|
|
1789
|
+
// finding-driven rollback keeps earlier passes — the findings channel
|
|
1790
|
+
// itself re-examines the repaired work next round (stable ids).
|
|
1791
|
+
if (rolledBack.length > 0) invalidateChecksForRolledBackTasks(ex.items, ex.tasks, rolledBack);
|
|
1792
|
+
const knownIds = new Set(coveredTaskIds);
|
|
1793
|
+
const mappedHighIds = [...new Set(actionableHighs.flatMap((h) => h.taskIds).filter((id) => knownIds.has(id)))];
|
|
1794
|
+
const highRolledBack = findingsRollbackSet(ex.tasks, mappedHighIds);
|
|
1795
|
+
const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
|
|
1796
|
+
const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
|
|
1797
|
+
const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
|
|
1798
|
+
// v0.9.2: capture the round's rollback set HERE. The reopen helpers above
|
|
1799
|
+
// mutate the tree and report only flipped nodes, so this is the only
|
|
1800
|
+
// moment the authoritative provenance exists; the still-open subset is
|
|
1801
|
+
// what blocks the next round and what wakes/pauses name explicitly.
|
|
1802
|
+
ex.blocked = allRolledBack.length > 0
|
|
1803
|
+
? {
|
|
1804
|
+
rolledBack: [...allRolledBack],
|
|
1805
|
+
tasks: blockedReviewTasks(ex.tasks, allRolledBack),
|
|
1806
|
+
round: ex.audit.rounds,
|
|
1807
|
+
escalatedRounds: 0,
|
|
1808
|
+
since: utcNow(),
|
|
1809
|
+
}
|
|
1810
|
+
: null;
|
|
1811
|
+
ex.blockedWakeTasks = undefined;
|
|
1812
|
+
withExecutionCheckpoint(ctx, (cp) => {
|
|
1813
|
+
// v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
|
|
1814
|
+
// re-stamp the checkpoint's plan identity in the same revision so a
|
|
1815
|
+
// later /resume-plans accepts the amended plan instead of rejecting
|
|
1816
|
+
// it as plan-mismatch (which would cost a full re-approval).
|
|
1817
|
+
const amendedCp = amended.length > 0
|
|
1818
|
+
? applyExecutionPlanAmended(cp, planIdentityOf(ex.planPath, cp.plan?.version ?? 1), ex.audit.rounds)
|
|
1819
|
+
: cp;
|
|
1820
|
+
return applyExecutionProgress(amendedCp, {
|
|
1821
|
+
tasks: taskProgressMap(ex.tasks),
|
|
1822
|
+
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1823
|
+
blocked: ex.blocked,
|
|
1824
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1825
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1826
|
+
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
|
|
1827
|
+
});
|
|
1828
|
+
});
|
|
1829
|
+
if (allRolledBack.length > 0 || amended.length > 0) {
|
|
1830
|
+
// Rolling back (or appending) is itself forward progress for the
|
|
1831
|
+
// watchdog, but NOT for audit.rounds: that counter stays monotonic
|
|
1832
|
+
// so the repair loop is bounded. The repair belongs to the
|
|
1833
|
+
// executor — back to executing.
|
|
1834
|
+
ex.stall.rounds = 0;
|
|
1835
|
+
ex.stall.lastSnapshot = stallSnapshot();
|
|
1836
|
+
setRunStatusForReview(ctx, "executing");
|
|
1837
|
+
}
|
|
1838
|
+
persist(ctx);
|
|
1839
|
+
updateStatusWidget(ctx);
|
|
1840
|
+
// v0.8 wake, generalized (v0.9): per-round one-shot token — exactly one
|
|
1841
|
+
// triggerTurn per committed fix-needing outcome, even when the round
|
|
1842
|
+
// resolves long after the settle that spawned it.
|
|
1286
1843
|
if (!round.wakeSent) {
|
|
1287
1844
|
round.wakeSent = true;
|
|
1288
1845
|
const openTasks = flattenTaskViews(ex.tasks)
|
|
1289
1846
|
.filter((task) => !taskIsTerminal(task))
|
|
1290
1847
|
.map((task) => task.id);
|
|
1291
|
-
const stranded =
|
|
1292
|
-
const
|
|
1848
|
+
const stranded = allRolledBack.length === 0 && amended.length === 0;
|
|
1849
|
+
const highLines = actionableHighs.map((h) => `- ${h.id}${h.taskIds.length ? ` (${h.taskIds.join(", ")})` : ""}: ${h.note}`).join("\n");
|
|
1850
|
+
const reportRef = reportPath
|
|
1851
|
+
? `Full round report: ${reportPath}`
|
|
1852
|
+
: `Full round report (run dir unwritable — inline):\n\n---\n${reportText.slice(0, 4000)}`;
|
|
1853
|
+
// v0.9.1 (F-004): a pure VC-fail round keeps the v0.8 lead — never
|
|
1854
|
+
// announce "0 high-severity findings" over an empty block.
|
|
1855
|
+
const findingsLead = actionableHighs.length > 0
|
|
1856
|
+
? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
|
|
1857
|
+
: `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
|
|
1858
|
+
const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
|
|
1859
|
+
const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${budgetSpent(ex) ? (activeBudget(ex) === "unlimited" ? ` This was round ${ex.reviewRoundsTotal} against the unlimited budget's ${unlimitedHardCapCeiling(budgetCounters(ex))}-round safety cap: the next terminal-task cycle pauses the run for review.` : ` This was round ${ex.audit.rounds} of ${budgetLabel(ex)}: the next terminal-task cycle pauses the run for review (or completes if every check passed).`) : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
|
|
1293
1860
|
messaging().sendMessage(
|
|
1294
1861
|
{
|
|
1295
1862
|
customType: "pi-plans-audit-failed",
|
|
1296
|
-
content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}
|
|
1863
|
+
content: `${content}\n\nStill open: ${openTasks.join(", ") || "(none — the tree is terminal)"}`,
|
|
1297
1864
|
display: true,
|
|
1298
1865
|
},
|
|
1299
1866
|
{ triggerTurn: true },
|
|
@@ -1301,8 +1868,48 @@ async function commitReviewOutcome(
|
|
|
1301
1868
|
}
|
|
1302
1869
|
return; // The agent repairs; the next settle re-enters the loop.
|
|
1303
1870
|
}
|
|
1871
|
+
|
|
1872
|
+
// Fail-closed completion: every pending check affirmatively passed AND no
|
|
1873
|
+
// high finding remains. The highs guard is explicit (v0.9.1, F-001): a
|
|
1874
|
+
// no-report round skips the fix branch above, so this is the last line
|
|
1875
|
+
// against vacuously completing an empty-pendingIds round with an
|
|
1876
|
+
// unresolved high. An all-undeterminable round yields failed === [] —
|
|
1877
|
+
// completing here would be fail-open, marking a run done with nothing
|
|
1878
|
+
// verified.
|
|
1879
|
+
if (passed.length === pendingIds.length && highs.length === 0) {
|
|
1880
|
+
ex.audit.failed = [];
|
|
1881
|
+
ex.audit.undeterminable = [];
|
|
1882
|
+
// v0.9.2: a passing audit ends the blocker — the review consumed it.
|
|
1883
|
+
ex.blocked = null;
|
|
1884
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
1885
|
+
applyExecutionProgress(cp, {
|
|
1886
|
+
tasks: taskProgressMap(ex.tasks),
|
|
1887
|
+
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1888
|
+
blocked: null,
|
|
1889
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1890
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1891
|
+
audit: { rounds: ex.audit.rounds, passed: true, findings },
|
|
1892
|
+
}),
|
|
1893
|
+
);
|
|
1894
|
+
await completeExecution(ctx);
|
|
1895
|
+
return;
|
|
1896
|
+
}
|
|
1897
|
+
|
|
1304
1898
|
// Undeterminable-only round: the loop self-schedules the retry (Q-B) — no
|
|
1305
1899
|
// wake, no message; the dashboard/overlay carries the round counter.
|
|
1900
|
+
ex.audit.failed = failed;
|
|
1901
|
+
ex.audit.undeterminable = undeterminable;
|
|
1902
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
1903
|
+
applyExecutionProgress(cp, {
|
|
1904
|
+
tasks: taskProgressMap(ex.tasks),
|
|
1905
|
+
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1906
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1907
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1908
|
+
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
|
|
1909
|
+
}),
|
|
1910
|
+
);
|
|
1911
|
+
persist(ctx);
|
|
1912
|
+
updateStatusWidget(ctx);
|
|
1306
1913
|
await maybeContinueReview(ctx, ex);
|
|
1307
1914
|
}
|
|
1308
1915
|
|
|
@@ -1310,10 +1917,11 @@ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promis
|
|
|
1310
1917
|
if (execution !== ex) return;
|
|
1311
1918
|
if (!reviewOwed(ex)) return;
|
|
1312
1919
|
if (ex.stall.paused) return;
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1920
|
+
// The budget gate owns every pause (numeric exhaustion, the unlimited hard
|
|
1921
|
+
// cap, and the unlimited no-progress valve). An exhausted budget with every
|
|
1922
|
+
// check satisfied has already completed inside commitReviewOutcome; here
|
|
1923
|
+
// the review is still owed, so a spent budget pauses.
|
|
1924
|
+
if (!reviewBudgetGate(ctx, ex)) return;
|
|
1317
1925
|
await startReviewRound(ctx);
|
|
1318
1926
|
}
|
|
1319
1927
|
|
|
@@ -1899,7 +2507,10 @@ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
|
|
|
1899
2507
|
// This also bounds the zero-input continue loop in agent_before_settle.
|
|
1900
2508
|
if (ex.stall.paused) return false;
|
|
1901
2509
|
if (ex.review.inFlight) return false;
|
|
1902
|
-
|
|
2510
|
+
// The budget panel is open: the review is being resolved, not owed anew.
|
|
2511
|
+
if (ex.review.budgetAsking) return false;
|
|
2512
|
+
return allTasksTerminal(ex.tasks)
|
|
2513
|
+
&& (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
|
|
1903
2514
|
}
|
|
1904
2515
|
|
|
1905
2516
|
/** v0.7.1: record that this settle already ran (or declined) its audit, so a
|
|
@@ -1984,6 +2595,33 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
|
|
|
1984
2595
|
if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
|
|
1985
2596
|
runtime.handled = true;
|
|
1986
2597
|
const ex = runtime.owner;
|
|
2598
|
+
// v0.9.2: while a failed round's tasks are still open, the next round
|
|
2599
|
+
// cannot start — and this ladder must NOT ride `stall.rounds`, which any
|
|
2600
|
+
// successful tool call resets: an executor that investigates but never
|
|
2601
|
+
// closes the reopened tasks has to escalate (and eventually pause) anyway.
|
|
2602
|
+
const blockedIds = blockedTaskIds(ex);
|
|
2603
|
+
if (ex.blocked && blockedIds.length > 0) {
|
|
2604
|
+
// Progress = fewer blockers than at the PREVIOUS blocked wake. The
|
|
2605
|
+
// stored `blocked.tasks` cannot serve as that baseline: every task
|
|
2606
|
+
// close re-syncs it, so it always equals the live set and the ladder
|
|
2607
|
+
// would only ever increment (review round 1, F-001). A same-size but
|
|
2608
|
+
// different set counts as no progress; a new member counts as none.
|
|
2609
|
+
const baseline = ex.blockedWakeTasks;
|
|
2610
|
+
const progressed = baseline !== undefined && blockedIds.length < baseline.length;
|
|
2611
|
+
ex.blockedWakeTasks = [...blockedIds];
|
|
2612
|
+
ex.blocked.escalatedRounds = progressed ? 0 : ex.blocked.escalatedRounds + 1;
|
|
2613
|
+
ex.blocked.tasks = blockedIds;
|
|
2614
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { blocked: ex.blocked }));
|
|
2615
|
+
if (ex.blocked.escalatedRounds >= STALL_MAX_ROUNDS) {
|
|
2616
|
+
pauseForStall(ctx, blockedPauseReason(ex));
|
|
2617
|
+
return;
|
|
2618
|
+
}
|
|
2619
|
+
notifyBlockedEscalation(ctx, ex);
|
|
2620
|
+
persist(ctx);
|
|
2621
|
+
updateStatusWidget(ctx);
|
|
2622
|
+
sendContinuationWake(ctx, runtime);
|
|
2623
|
+
return;
|
|
2624
|
+
}
|
|
1987
2625
|
const snapshot = stallSnapshot();
|
|
1988
2626
|
const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
|
|
1989
2627
|
ex.stall.lastSnapshot = snapshot;
|
|
@@ -2003,8 +2641,10 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
|
|
|
2003
2641
|
|
|
2004
2642
|
export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
|
|
2005
2643
|
// v0.6.1: continuation wakes are one-shot; stale ones (including the
|
|
2006
|
-
// legacy v0.6.0 goal-wait type) never replay after a restart.
|
|
2644
|
+
// legacy v0.6.0 goal-wait type) never replay after a restart. v0.9.2: the
|
|
2645
|
+
// visible escalated-blocked line is one-shot for the same reason.
|
|
2007
2646
|
return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
|
|
2647
|
+
&& message.customType !== EXECUTION_BLOCKED_CUSTOM_TYPE
|
|
2008
2648
|
&& message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
|
|
2009
2649
|
}
|
|
2010
2650
|
|
|
@@ -2013,18 +2653,20 @@ export function filterContinuationMessages<T extends { customType?: string; deta
|
|
|
2013
2653
|
}
|
|
2014
2654
|
|
|
2015
2655
|
/** Called for genuine user input or an explicit same-execution resume.
|
|
2016
|
-
* v0.8: a REVIEW
|
|
2017
|
-
* refill the
|
|
2656
|
+
* v0.8: a REVIEW pause is never lifted here — ordinary input must not
|
|
2657
|
+
* refill the review budget (CF2-004); only /plans-execute
|
|
2018
2658
|
* (resumeActiveExecution) is the explicit confirmation surface. Genuine
|
|
2019
2659
|
* stall pauses still clear on input as before. */
|
|
2020
2660
|
export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
2021
2661
|
const ex = getExecution();
|
|
2022
2662
|
if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
|
|
2023
|
-
if (
|
|
2663
|
+
if (isReviewPauseReason(ex.stall.pausedReason)) {
|
|
2024
2664
|
// Surfaced once per input so the user is not left guessing why the run
|
|
2025
2665
|
// stays paused; the pause itself and the budget survive untouched.
|
|
2666
|
+
// v0.9.3: the note names the budget actually in force and the fact that
|
|
2667
|
+
// the confirmation re-opens the picker.
|
|
2026
2668
|
ctx.ui.notify?.(
|
|
2027
|
-
|
|
2669
|
+
`pi-plans: the review budget is exhausted (${budgetLabel(ex)}) — run /plans-execute to grant a fresh budget (it re-opens the round-count picker; that confirmation is the only surface that does).`,
|
|
2028
2670
|
"warning",
|
|
2029
2671
|
);
|
|
2030
2672
|
return false;
|
|
@@ -2033,18 +2675,61 @@ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
|
2033
2675
|
ex.stall.pausedReason = undefined;
|
|
2034
2676
|
ex.stall.rounds = 0;
|
|
2035
2677
|
ex.stall.lastSnapshot = stallSnapshot();
|
|
2036
|
-
|
|
2678
|
+
// v0.9.2: a resume grants a fresh escalation ladder (the blocker itself is
|
|
2679
|
+
// still recorded, so the next wake names it) and never a fresh review
|
|
2680
|
+
// budget — that stays `/plans-execute`-only.
|
|
2681
|
+
if (ex.blocked) {
|
|
2682
|
+
ex.blocked.escalatedRounds = 0;
|
|
2683
|
+
ex.blockedWakeTasks = undefined;
|
|
2684
|
+
}
|
|
2685
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null, blocked: ex.blocked }));
|
|
2037
2686
|
persist(ctx);
|
|
2038
2687
|
updateStatusWidget(ctx);
|
|
2039
2688
|
return true;
|
|
2040
2689
|
}
|
|
2041
2690
|
|
|
2042
|
-
|
|
2043
|
-
|
|
2044
|
-
|
|
2045
|
-
|
|
2691
|
+
/** Outcome of an explicit `/plans-execute` resume, so the calling tool can
|
|
2692
|
+
* report what actually happened (v0.9.3: the grant may re-open the picker). */
|
|
2693
|
+
export interface ResumeOutcome {
|
|
2694
|
+
resumed: boolean;
|
|
2695
|
+
/** Set when the pause was lifted by a budget grant. */
|
|
2696
|
+
grantedBudget?: ReviewBudget;
|
|
2697
|
+
/** True when the user closed the budget panel — the pause stands. */
|
|
2698
|
+
budgetDeclined?: boolean;
|
|
2699
|
+
}
|
|
2700
|
+
|
|
2701
|
+
export async function resumeActiveExecution(ctx: ExtensionContext): Promise<ResumeOutcome> {
|
|
2702
|
+
// v0.8: /plans-execute is THE explicit confirmation surface for a review
|
|
2703
|
+
// pause — the only place a fresh budget is granted (Q-confirm-surface).
|
|
2704
|
+
// Ordinary input and session restores never refill.
|
|
2705
|
+
// v0.9.3: the grant re-opens the budget picker with the current value
|
|
2706
|
+
// preselected; Esc keeps the run paused (an explicit confirmation is the
|
|
2707
|
+
// only way forward). Headless sessions, which have no panel to show, keep
|
|
2708
|
+
// the current budget instead of stranding the run.
|
|
2046
2709
|
const pausedEx = getExecution();
|
|
2047
|
-
if (pausedEx?.stall.paused &&
|
|
2710
|
+
if (pausedEx?.stall.paused && isReviewPauseReason(pausedEx.stall.pausedReason)) {
|
|
2711
|
+
const previous = activeBudget(pausedEx);
|
|
2712
|
+
let granted = previous;
|
|
2713
|
+
if (reviewBudgetPanelUsable(ctx)) {
|
|
2714
|
+
const picked = await askReviewBudget(ctx, pausedEx.uiLanguage, previous);
|
|
2715
|
+
if (picked === null) {
|
|
2716
|
+
messaging().sendMessage(
|
|
2717
|
+
{
|
|
2718
|
+
customType: "pi-plans-review-budget-declined",
|
|
2719
|
+
content: `**pi-plans: review budget unchanged (${formatReviewBudget(previous)})** — the run stays paused. Run /plans-execute and pick a round count to continue.`,
|
|
2720
|
+
display: true,
|
|
2721
|
+
},
|
|
2722
|
+
{ triggerTurn: false },
|
|
2723
|
+
);
|
|
2724
|
+
return { resumed: false, budgetDeclined: true };
|
|
2725
|
+
}
|
|
2726
|
+
granted = picked;
|
|
2727
|
+
// Q-4: only a grant that lands on `unlimited` lifts the hard cap —
|
|
2728
|
+
// the cumulative counter itself never resets.
|
|
2729
|
+
if (granted === "unlimited") {
|
|
2730
|
+
pausedEx.reviewCapExtension += UNLIMITED_HARD_CAP;
|
|
2731
|
+
}
|
|
2732
|
+
}
|
|
2048
2733
|
pausedEx.stall.paused = false;
|
|
2049
2734
|
pausedEx.stall.pausedReason = undefined;
|
|
2050
2735
|
pausedEx.stall.rounds = 0;
|
|
@@ -2052,26 +2737,45 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
|
2052
2737
|
pausedEx.audit.rounds = 0;
|
|
2053
2738
|
pausedEx.audit.failed = [];
|
|
2054
2739
|
pausedEx.audit.undeterminable = [];
|
|
2740
|
+
// v0.9.3: a fresh budget window also resets the no-progress valve.
|
|
2741
|
+
pausedEx.reviewNoProgress = undefined;
|
|
2742
|
+
// v0.9.2: the explicit confirmation also refreshes the blocked ladder
|
|
2743
|
+
// (the blocker set itself survives — it still names what stays open).
|
|
2744
|
+
if (pausedEx.blocked) {
|
|
2745
|
+
pausedEx.blocked.escalatedRounds = 0;
|
|
2746
|
+
pausedEx.blockedWakeTasks = undefined;
|
|
2747
|
+
}
|
|
2748
|
+
// v0.9: the fresh budget inherits unresolved findings (stable ids keep
|
|
2749
|
+
// counting) — only the round counter resets.
|
|
2055
2750
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
2056
2751
|
applyExecutionProgress(cp, {
|
|
2057
2752
|
tasks: taskProgressMap(pausedEx.tasks),
|
|
2058
|
-
audit: { rounds: 0, lastResult: undefined },
|
|
2753
|
+
audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
|
|
2754
|
+
blocked: pausedEx.blocked,
|
|
2755
|
+
reviewBudget: granted,
|
|
2756
|
+
reviewBudgetDefaulted: pausedEx.reviewBudgetDefaulted === true,
|
|
2757
|
+
reviewRoundsTotal: pausedEx.reviewRoundsTotal,
|
|
2758
|
+
reviewCapExtension: pausedEx.reviewCapExtension,
|
|
2759
|
+
reviewNoProgress: null,
|
|
2059
2760
|
pausedReason: null,
|
|
2060
2761
|
}),
|
|
2061
2762
|
);
|
|
2062
2763
|
persist(ctx);
|
|
2063
2764
|
updateStatusWidget(ctx);
|
|
2765
|
+
const grantedNote = granted === "unlimited"
|
|
2766
|
+
? `unlimited review budget granted (hard cap now ${unlimitedHardCapCeiling(budgetCounters(pausedEx))} rounds; the run has spent ${pausedEx.reviewRoundsTotal})`
|
|
2767
|
+
: `fresh ${formatReviewBudget(granted)}-round review budget granted`;
|
|
2064
2768
|
messaging().sendMessage(
|
|
2065
2769
|
{
|
|
2066
2770
|
customType: "pi-plans-review-budget-granted",
|
|
2067
|
-
content:
|
|
2771
|
+
content: `**pi-plans: ${grantedNote}** — the execution review resumes now.`,
|
|
2068
2772
|
display: true,
|
|
2069
2773
|
},
|
|
2070
2774
|
{ triggerTurn: false },
|
|
2071
2775
|
);
|
|
2072
2776
|
const grantChain = launchReviewRound(ctx);
|
|
2073
2777
|
if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
|
|
2074
|
-
return true;
|
|
2778
|
+
return { resumed: true, grantedBudget: granted };
|
|
2075
2779
|
}
|
|
2076
2780
|
// v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
|
|
2077
2781
|
// to `return false` here, so `/plans-execute` answered "already executing"
|
|
@@ -2082,15 +2786,15 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
|
2082
2786
|
latchAuditThisSettle();
|
|
2083
2787
|
const chain = launchReviewRound(ctx);
|
|
2084
2788
|
if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
|
|
2085
|
-
return true;
|
|
2789
|
+
return { resumed: true };
|
|
2086
2790
|
}
|
|
2087
|
-
if (!resumeGoalWaitIfPaused(ctx)) return false;
|
|
2791
|
+
if (!resumeGoalWaitIfPaused(ctx)) return { resumed: false };
|
|
2088
2792
|
const runtime = currentContinuationRuntime(ctx)!;
|
|
2089
2793
|
if (canWakeExecution(ctx, runtime)) {
|
|
2090
2794
|
runtime.handled = true;
|
|
2091
2795
|
sendContinuationWake(ctx, runtime);
|
|
2092
2796
|
}
|
|
2093
|
-
return true;
|
|
2797
|
+
return { resumed: true };
|
|
2094
2798
|
}
|
|
2095
2799
|
|
|
2096
2800
|
export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
@@ -2102,6 +2806,19 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
2102
2806
|
const summary = flat
|
|
2103
2807
|
.map((task) => `- ${taskIsTerminal(task) && task.status === "skipped" ? "~" : "✓"} \`${task.id}\` ${task.title}`)
|
|
2104
2808
|
.join("\n");
|
|
2809
|
+
// v0.9: residual (non-high) findings are summarized, never silent — the
|
|
2810
|
+
// loop converged because nothing high-blocking remained.
|
|
2811
|
+
const residualFindings = execution.audit.findings.filter((f) => f.severity !== "high");
|
|
2812
|
+
const residualNote = residualFindings.length > 0
|
|
2813
|
+
? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
|
|
2814
|
+
: "";
|
|
2815
|
+
// v0.9.3 (Q-2, round-1 F-004): an exhausted budget may complete WITH
|
|
2816
|
+
// unresolved high findings — the completion surface must say so instead of
|
|
2817
|
+
// claiming "execution review passed".
|
|
2818
|
+
const toleratedHighs = execution.audit.findings.filter((f) => f.severity === "high");
|
|
2819
|
+
const toleratedNote = toleratedHighs.length > 0
|
|
2820
|
+
? `\n\n⚠️ The review budget was exhausted (${budgetLabel(execution)}) before these high-severity finding(s) could be resolved: ${toleratedHighs.map((f) => f.id).join(", ")} — every verification check passed; the finding(s) remain in the round reports under the run directory.`
|
|
2821
|
+
: "";
|
|
2105
2822
|
const planPath = execution.planPath;
|
|
2106
2823
|
withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
|
|
2107
2824
|
execution = null;
|
|
@@ -2111,7 +2828,9 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
2111
2828
|
messaging().sendMessage(
|
|
2112
2829
|
{
|
|
2113
2830
|
customType: "pi-plans-complete",
|
|
2114
|
-
content:
|
|
2831
|
+
content: toleratedHighs.length > 0
|
|
2832
|
+
? `**Plan complete (review budget exhausted).** ⚠️ \`${planPath}\` — all verification checks passed; ${toleratedHighs.length} high-severity finding(s) stayed unresolved.\n\n${summary}${residualNote}${toleratedNote}`
|
|
2833
|
+
: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
|
|
2115
2834
|
display: true,
|
|
2116
2835
|
},
|
|
2117
2836
|
{ triggerTurn: false },
|
|
@@ -2146,13 +2865,32 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
|
|
|
2146
2865
|
const rollbackNote = execution.audit.failed.length > 0
|
|
2147
2866
|
? `\nExecution review round ${execution.audit.rounds} failed checks: ${execution.audit.failed.join(", ")} — the covered tasks were rolled back to pending; re-close them with evidence after fixing the failures.`
|
|
2148
2867
|
: "";
|
|
2868
|
+
const unresolvedHighs = unresolvedHighFindings(execution);
|
|
2869
|
+
const highFindingsNote = unresolvedHighs.length > 0
|
|
2870
|
+
? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
|
|
2871
|
+
: "";
|
|
2872
|
+
// v0.9.2: the wake must answer "is the review running?" explicitly. The
|
|
2873
|
+
// previous text left the executor waiting for a round that cannot start
|
|
2874
|
+
// while the tree is open (the exact stall this feature fixes).
|
|
2875
|
+
const blockedIds = blockedTaskIds(execution);
|
|
2876
|
+
const blockedRound = execution.blocked?.round ?? execution.audit.rounds;
|
|
2877
|
+
const outstanding = reviewOutstanding(execution);
|
|
2878
|
+
const reviewLine = execution.review.inFlight
|
|
2879
|
+
? `\nReview: round ${execution.audit.rounds + 1}/${budgetLabel(execution)} IS RUNNING (read-only reviewer verifying) — do not wait on it and do not re-close tasks for it.`
|
|
2880
|
+
: blockedIds.length > 0 && outstanding
|
|
2881
|
+
? `\nReview: NO round is running — the task tree is not terminal, so the review cannot start. It starts by itself the moment every task is terminal (round ${execution.audit.rounds + 1}/${budgetLabel(execution)}).`
|
|
2882
|
+
: "";
|
|
2883
|
+
const escalated = (execution.blocked?.escalatedRounds ?? 0) > 0;
|
|
2884
|
+
const blockedLine = blockedIds.length > 0 && outstanding
|
|
2885
|
+
? `\nBLOCKED — ${blockedIds.length} task(s) reopened by round ${blockedRound} are still open:\n${blockedIds.map((id) => `- ${id} (reopened by round ${blockedRound}) — close with \`plans_update_task\`: status "complete" with evidence, or "skipped" with a skipReason. Closing a child does NOT close its parent; a parent with an open child is not terminal.`).join("\n")}${escalated ? `\nThis is blocked wake ${execution.blocked?.escalatedRounds}/${STALL_MAX_ROUNDS}: if these tasks stay open, the watchdog pauses the run and a human has to resume it.` : ""}`
|
|
2886
|
+
: "";
|
|
2149
2887
|
return `[PI-PLANS EXECUTION — write access enabled]
|
|
2150
2888
|
Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
|
|
2151
2889
|
|
|
2152
2890
|
Current wave ${currentWave} open tasks:
|
|
2153
2891
|
${waveList}
|
|
2154
2892
|
|
|
2155
|
-
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}
|
|
2893
|
+
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}${reviewLine}${blockedLine}
|
|
2156
2894
|
|
|
2157
2895
|
${graphLine}
|
|
2158
2896
|
|
|
@@ -2160,6 +2898,8 @@ Execution rules:
|
|
|
2160
2898
|
- Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
|
|
2161
2899
|
- Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
|
|
2162
2900
|
- Close subtasks before their parent; a parent is auditable only when every child is terminal.
|
|
2901
|
+
- After a FAILED review round, every reopened task must be re-closed with fresh evidence — PARENTS INCLUDED. Closing a task's children does NOT close the task: a parent that still has an open child is not terminal, and the review cannot start until the whole tree is terminal.
|
|
2902
|
+
- NEVER wait for the review. While any task is open, the review is not running and nothing will re-close tasks for you: an open task is your work queue — close it with evidence or skip it with a skipReason.
|
|
2163
2903
|
- When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
|
|
2164
2904
|
- Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
|
|
2165
2905
|
- Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
|
|
@@ -2217,7 +2957,12 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
2217
2957
|
planTasks = parsePlanTasks(planText);
|
|
2218
2958
|
const snapshotProgress = taskProgressMap(snapshot.tasks ?? []);
|
|
2219
2959
|
tasks = buildTaskView(planTasks, snapshotProgress);
|
|
2220
|
-
|
|
2960
|
+
// Fresh text wins, but the satisfied state survives the re-parse: a
|
|
2961
|
+
// mid-loop session restore (v0.9.1, found via F-001's self-schedule
|
|
2962
|
+
// path) must not hand the next round a brief that re-judges checks an
|
|
2963
|
+
// earlier round already passed — that burned budget on every /reload.
|
|
2964
|
+
const snapDone = new Set(snapshot.items.filter((c) => c.done).map((c) => c.id));
|
|
2965
|
+
items = parseChecklist(planText).map((item) => (snapDone.has(item.id) ? { ...item, done: true } : item));
|
|
2221
2966
|
} catch {
|
|
2222
2967
|
tasks = snapshot.tasks;
|
|
2223
2968
|
}
|
|
@@ -2234,14 +2979,27 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
2234
2979
|
usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
|
|
2235
2980
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
2236
2981
|
stall: { ...snapshot.stall, lastSnapshot: null },
|
|
2982
|
+
// v0.9.2: a snapshot taken by a pre-feature build carries the live failed
|
|
2983
|
+
// set but no blocker record; reconstruct it so the very first wake after
|
|
2984
|
+
// a /reload (the recovery path for an already-stuck run) names the tasks.
|
|
2985
|
+
blocked: snapshot.blocked ?? blockedFromFailedIds(snapshot.audit?.failed ?? [], snapshot.audit?.rounds ?? 0, tasks, items),
|
|
2237
2986
|
audit: {
|
|
2238
2987
|
rounds: snapshot.audit?.rounds ?? 0,
|
|
2239
2988
|
failed: snapshot.audit?.failed ?? [],
|
|
2240
2989
|
undeterminable: [],
|
|
2990
|
+
findings: toReviewFindings(snapshot.audit?.findings),
|
|
2241
2991
|
running: false,
|
|
2242
2992
|
},
|
|
2243
|
-
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
2993
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
2244
2994
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
2995
|
+
// v0.9.3 (round-1 F-005): the snapshot's budget/counters survive a
|
|
2996
|
+
// /reload — an unset budget stays unset (the picker runs before round 1),
|
|
2997
|
+
// a legacy snapshot resolves to the 5-round bound via the audit counter.
|
|
2998
|
+
reviewBudget: resolveStoredBudget(snapshot.reviewBudget, snapshot.audit?.rounds ?? 0),
|
|
2999
|
+
reviewBudgetDefaulted: snapshot.reviewBudgetDefaulted === true,
|
|
3000
|
+
reviewRoundsTotal: snapshot.reviewRoundsTotal ?? 0,
|
|
3001
|
+
reviewCapExtension: snapshot.reviewCapExtension ?? 0,
|
|
3002
|
+
reviewNoProgress: snapshot.reviewNoProgress ?? undefined,
|
|
2245
3003
|
};
|
|
2246
3004
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
2247
3005
|
resetContinuationRuntime(ctx);
|