pi-plans 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +13 -6
- package/references/state-and-config.md +1 -1
- package/src/auditor.ts +12 -5
- package/src/dashboard.ts +60 -9
- package/src/exec.ts +596 -43
- package/src/resume-command.ts +20 -6
- package/src/review-budget.ts +290 -0
- package/src/tasks.ts +53 -0
- package/src/ui-language.ts +38 -0
- package/src/workflow-state.ts +138 -3
- package/tests/dashboard.test.ts +98 -0
- package/tests/exec-review-loop.test.ts +632 -4
- package/tests/exec.test.ts +32 -5
- package/tests/fixtures/lattice-code-blocked/plan-v2-trimmed.md +57 -0
- package/tests/fixtures/lattice-code-blocked/state.json +158 -0
- package/tests/resume.test.ts +45 -1
- package/tests/review-budget.test.ts +201 -0
- package/tests/tasks.test.ts +45 -0
- package/tests/workflow-state.test.ts +187 -0
- package/tools/execute-plan.ts +12 -5
package/src/exec.ts
CHANGED
|
@@ -66,6 +66,8 @@ import {
|
|
|
66
66
|
sha256File,
|
|
67
67
|
StaleCheckpointError,
|
|
68
68
|
type ExecutionApproval,
|
|
69
|
+
type ExecutionBlocked,
|
|
70
|
+
type ExecutionCheckpoint,
|
|
69
71
|
type WorkflowCheckpoint,
|
|
70
72
|
} from "./workflow-state.ts";
|
|
71
73
|
import { graphBlockForExecutor } from "./code-graph/prompts.ts";
|
|
@@ -82,12 +84,14 @@ import {
|
|
|
82
84
|
allTasksTerminal,
|
|
83
85
|
auditRollbackSet,
|
|
84
86
|
auditableChecks,
|
|
87
|
+
blockedReviewTasks,
|
|
85
88
|
buildTaskView,
|
|
86
89
|
currentTask,
|
|
87
90
|
findingsRollbackSet,
|
|
88
91
|
flattenTaskViews,
|
|
89
92
|
invalidateChecksForRolledBackTasks,
|
|
90
93
|
maxWave,
|
|
94
|
+
rollbackCoverageIds,
|
|
91
95
|
taskIsTerminal,
|
|
92
96
|
taskProgress,
|
|
93
97
|
taskProgressMap,
|
|
@@ -103,7 +107,27 @@ import {
|
|
|
103
107
|
renderDashboardLines,
|
|
104
108
|
renderDashboardTreeLines,
|
|
105
109
|
} from "./dashboard.ts";
|
|
106
|
-
import {
|
|
110
|
+
import { presolvedCheckIds, runCompletionAudit, writeReviewRoundReport, type AuditOutcome, type AuditRoundResult, type ReviewFinding } from "./auditor.ts";
|
|
111
|
+
import {
|
|
112
|
+
DEFAULT_REVIEW_BUDGET,
|
|
113
|
+
LEGACY_REVIEW_MAX_ROUNDS,
|
|
114
|
+
NO_PROGRESS_MAX_STREAK,
|
|
115
|
+
REVIEW_CAP_PAUSE_PREFIX,
|
|
116
|
+
REVIEW_NO_PROGRESS_PAUSE_PREFIX,
|
|
117
|
+
UNLIMITED_HARD_CAP,
|
|
118
|
+
askReviewBudget,
|
|
119
|
+
budgetExhausted,
|
|
120
|
+
bumpNoProgress,
|
|
121
|
+
formatReviewBudget,
|
|
122
|
+
isReviewPauseReason,
|
|
123
|
+
noProgressSignature,
|
|
124
|
+
noProgressTripped,
|
|
125
|
+
resolveStoredBudget,
|
|
126
|
+
reviewBudgetPanelAvailable,
|
|
127
|
+
unlimitedHardCapCeiling,
|
|
128
|
+
type NoProgressState,
|
|
129
|
+
type ReviewBudget,
|
|
130
|
+
} from "./review-budget.ts";
|
|
107
131
|
import type { ReviewFindingRecord } from "./workflow-state.ts";
|
|
108
132
|
import { staleReloadHint as probeStaleReload } from "./staleness.ts";
|
|
109
133
|
import { messaging } from "./messaging.ts";
|
|
@@ -137,13 +161,48 @@ export interface ExecState {
|
|
|
137
161
|
/** Execution-review loop (v0.8), memory-only: the attempt index names the
|
|
138
162
|
* per-round report files; consecutiveDiscards bounds the fingerprint
|
|
139
163
|
* re-run loop; inFlight owns the round's abort lifecycle. */
|
|
140
|
-
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null };
|
|
164
|
+
review: { attempts: number; consecutiveDiscards: number; inFlight: InFlightReview | null; budgetAsking?: boolean };
|
|
141
165
|
/** Per-settle audit latch (v0.7.1): a settled round fires the completion
|
|
142
166
|
* audit at most once, so the turn_end / agent_before_settle / resume entry
|
|
143
167
|
* points cannot double-consume a round when several land in one settle.
|
|
144
168
|
* Created on demand by auditLatchOf(); every construction path may omit it. */
|
|
145
169
|
auditLatch?: { auditedThisSettle: boolean; activity: number };
|
|
146
|
-
|
|
170
|
+
/** v0.9.2: outstanding blocker from the newest failed review round — the
|
|
171
|
+
* authoritative rollback set captured at commit time plus the still-open
|
|
172
|
+
* tasks among it. Memory mirror of `execution.blocked`; null = nothing
|
|
173
|
+
* blocks the review. */
|
|
174
|
+
blocked: ExecBlocked | null;
|
|
175
|
+
/** v0.9.2: blocker ids as observed at the PREVIOUS blocked wake — the
|
|
176
|
+
* ladder's progress baseline. Memory-only: `blocked.tasks` is re-synced by
|
|
177
|
+
* `persistTaskProgress` on every task close (so it stays fresh for the
|
|
178
|
+
* resume brief), which would make a comparison against it meaningless.
|
|
179
|
+
* Absent after a restore/resume → the next blocked wake restarts the count. */
|
|
180
|
+
blockedWakeTasks?: string[];
|
|
181
|
+
/** v0.9.2: highest escalation level already surfaced as a visible system
|
|
182
|
+
* line. Memory-only (never persisted/snapshotted) so a restored session
|
|
183
|
+
* re-notifies once. */
|
|
184
|
+
blockedNotifiedLevel?: number;
|
|
185
|
+
/** v0.9.3: the per-run execution-review budget. Resolved exactly once —
|
|
186
|
+
* right before round 1 — from the picker or the no-UI default, then
|
|
187
|
+
* persisted. `undefined` = not decided yet (never conflated with a legacy
|
|
188
|
+
* checkpoint, which `loadExecutionFromCheckpoint` resolves to 5). */
|
|
189
|
+
reviewBudget?: ReviewBudget;
|
|
190
|
+
/** v0.9.3: the budget came from the fallback (no UI/headless/auto-approve),
|
|
191
|
+
* not from a user pick — the expanded dashboard row marks it `(default)`
|
|
192
|
+
* and the resume brief repeats the note. Persisted so a later session can
|
|
193
|
+
* still tell the difference. */
|
|
194
|
+
reviewBudgetDefaulted?: boolean;
|
|
195
|
+
/** v0.9.3: committed review rounds across the whole run (never reset by a
|
|
196
|
+
* grant) — the unlimited budget's cumulative hard-cap counter. */
|
|
197
|
+
reviewRoundsTotal: number;
|
|
198
|
+
/** v0.9.3: rounds added to the unlimited hard cap by explicit grants. */
|
|
199
|
+
reviewCapExtension: number;
|
|
200
|
+
/** v0.9.3: no-progress valve state (unlimited budget only). */
|
|
201
|
+
reviewNoProgress?: NoProgressState;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** Live blocker record (see `ExecutionBlocked` in workflow-state.ts). */
|
|
205
|
+
type ExecBlocked = ExecutionBlocked;
|
|
147
206
|
|
|
148
207
|
/** One in-flight review round: owns its abort lifecycle, its fingerprint of
|
|
149
208
|
* the audited subject, and the per-round one-shot wake token (v0.8). */
|
|
@@ -173,6 +232,8 @@ const STALL_MAX_ROUNDS = 3;
|
|
|
173
232
|
let execution: ExecState | null = null;
|
|
174
233
|
|
|
175
234
|
export const EXECUTION_CONTINUE_CUSTOM_TYPE = "pi-plans-exec-continue";
|
|
235
|
+
/** v0.9.2: visible escalation line — never replays after a restart. */
|
|
236
|
+
export const EXECUTION_BLOCKED_CUSTOM_TYPE = "pi-plans-exec-blocked";
|
|
176
237
|
/** Legacy v0.6.0 continuation message type — filtered on restore. */
|
|
177
238
|
const LEGACY_GOAL_WAIT_CUSTOM_TYPE = "pi-plans-goal-wait";
|
|
178
239
|
|
|
@@ -239,9 +300,51 @@ export interface CheckpointExecutionLoad {
|
|
|
239
300
|
* so the /resume-plans brief can surface outstanding highs before the
|
|
240
301
|
* per-turn injection ever runs. */
|
|
241
302
|
findings?: ReviewFinding[];
|
|
303
|
+
/** v0.9.2: persisted blocker of the newest failed round, so the resume
|
|
304
|
+
* brief can name the open tasks that keep the review from starting. */
|
|
305
|
+
blocked?: ExecutionBlocked | null;
|
|
306
|
+
/** v0.9.3: the persisted per-run review budget (or undefined when the run
|
|
307
|
+
* never picked one — a legacy checkpoint resolves to 5 in the live state,
|
|
308
|
+
* see `reviewBudgetDefaulted`). */
|
|
309
|
+
reviewBudget?: ReviewBudget;
|
|
310
|
+
reviewBudgetDefaulted?: boolean;
|
|
311
|
+
reviewRoundsTotal?: number;
|
|
312
|
+
reviewCapExtension?: number;
|
|
242
313
|
error?: string;
|
|
243
314
|
}
|
|
244
315
|
|
|
316
|
+
/** v0.9.2: a checkpoint written before `execution.blocked` existed still names
|
|
317
|
+
* its blocker — `audit.lastResult` records the newest failed round's ids and
|
|
318
|
+
* the coverage cascade is pure, so the rollback set can be reconstructed
|
|
319
|
+
* WITHOUT re-applying anything (the tree already carries the reopen's outcome;
|
|
320
|
+
* re-applying it would revert tasks the executor has since re-closed).
|
|
321
|
+
* Finding-driven rounds (`highs: F-…`) carry no coverage and stay without a
|
|
322
|
+
* record — the wake still states that no round can start while tasks are open. */
|
|
323
|
+
function blockedFromFailedIds(
|
|
324
|
+
failedIds: readonly string[],
|
|
325
|
+
round: number,
|
|
326
|
+
tasks: TaskView[],
|
|
327
|
+
checklist: CheckItem[],
|
|
328
|
+
): ExecBlocked | null {
|
|
329
|
+
if (failedIds.length === 0) return null;
|
|
330
|
+
const rolledBack = rollbackCoverageIds(tasks, checklist, failedIds, []);
|
|
331
|
+
if (rolledBack.length === 0) return null;
|
|
332
|
+
const open = blockedReviewTasks(tasks, rolledBack);
|
|
333
|
+
if (open.length === 0) return null;
|
|
334
|
+
return { rolledBack, tasks: open, round, escalatedRounds: 0, since: utcNow() };
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
function backfillBlocked(
|
|
338
|
+
checkpoint: ExecutionCheckpoint,
|
|
339
|
+
tasks: TaskView[],
|
|
340
|
+
checklist: CheckItem[],
|
|
341
|
+
): ExecBlocked | null {
|
|
342
|
+
const last = checkpoint.audit?.lastResult?.trim() ?? "";
|
|
343
|
+
if (last.length === 0 || last.startsWith("highs:")) return null;
|
|
344
|
+
const failed = last.split(",").map((id) => id.trim()).filter((id) => id.length > 0);
|
|
345
|
+
return blockedFromFailedIds(failed, checkpoint.audit?.rounds ?? 0, tasks, checklist);
|
|
346
|
+
}
|
|
347
|
+
|
|
245
348
|
/**
|
|
246
349
|
* Shared restore primitive: load the executing state from a run checkpoint
|
|
247
350
|
* into THIS session. Authorization is kept only when the recorded approval
|
|
@@ -309,6 +412,11 @@ export function loadExecutionFromCheckpoint(
|
|
|
309
412
|
startedAt: utcNow(),
|
|
310
413
|
usage: { inToks: cp.execution.usage.inToks, outToks: cp.execution.usage.outToks },
|
|
311
414
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
415
|
+
// v0.9.2: the persisted blocker survives a restore so the resume brief
|
|
416
|
+
// and the dashboard can name it; a reverifyAll restore drops it together
|
|
417
|
+
// with the task progress it described. Pre-feature checkpoints are
|
|
418
|
+
// backfilled from the newest failed round's ids.
|
|
419
|
+
blocked: reverifyAll ? null : (cp.execution.blocked ?? backfillBlocked(cp.execution, tasks, items)),
|
|
312
420
|
// D-020: a paused legacy (or stopped) execution rebuilds unpaused — the
|
|
313
421
|
// resume itself is the user's intent; the reason is surfaced in the
|
|
314
422
|
// resume brief instead. EXCEPT a review-cap pause (v0.8): it must
|
|
@@ -327,8 +435,16 @@ export function loadExecutionFromCheckpoint(
|
|
|
327
435
|
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
328
436
|
running: false,
|
|
329
437
|
},
|
|
330
|
-
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
438
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
331
439
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
440
|
+
// v0.9.3: the review budget survives a restore; a checkpoint written
|
|
441
|
+
// before the feature that already spent rounds keeps the legacy 5-round
|
|
442
|
+
// bound instead of being cut to the new default mid-flight.
|
|
443
|
+
reviewBudget: resolveStoredBudget(cp.execution.reviewBudget, cp.execution.audit?.rounds ?? 0),
|
|
444
|
+
reviewBudgetDefaulted: cp.execution.reviewBudgetDefaulted === true,
|
|
445
|
+
reviewRoundsTotal: cp.execution.reviewRoundsTotal ?? 0,
|
|
446
|
+
reviewCapExtension: cp.execution.reviewCapExtension ?? 0,
|
|
447
|
+
reviewNoProgress: cp.execution.reviewNoProgress ?? undefined,
|
|
332
448
|
};
|
|
333
449
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
334
450
|
executionRunId = runId;
|
|
@@ -363,6 +479,11 @@ export function loadExecutionFromCheckpoint(
|
|
|
363
479
|
if (headChanged) {
|
|
364
480
|
withExecutionCheckpoint(ctx, (current) => applyExecutionHeadChanged(current));
|
|
365
481
|
}
|
|
482
|
+
// A backfilled blocker is authoritative from here on: persist it once so a
|
|
483
|
+
// later restart reads the record instead of re-deriving it.
|
|
484
|
+
if (!reverifyAll && (cp.execution.blocked ?? null) === null && execution!.blocked !== null) {
|
|
485
|
+
withExecutionCheckpoint(ctx, (current) => applyExecutionProgress(current, { blocked: execution!.blocked }));
|
|
486
|
+
}
|
|
366
487
|
persist(ctx);
|
|
367
488
|
updateStatusWidget(ctx);
|
|
368
489
|
return {
|
|
@@ -374,6 +495,13 @@ export function loadExecutionFromCheckpoint(
|
|
|
374
495
|
legacyPlan: planTasks.legacy,
|
|
375
496
|
legacyDelegate,
|
|
376
497
|
findings: toReviewFindings(cp.execution.audit?.findings),
|
|
498
|
+
// The LIVE record: a pre-feature checkpoint is backfilled (and persisted)
|
|
499
|
+
// above, so the resume brief sees the reconstructed blocker.
|
|
500
|
+
blocked: execution?.blocked ?? null,
|
|
501
|
+
reviewBudget: execution?.reviewBudget,
|
|
502
|
+
reviewBudgetDefaulted: execution?.reviewBudgetDefaulted === true,
|
|
503
|
+
reviewRoundsTotal: execution?.reviewRoundsTotal ?? 0,
|
|
504
|
+
reviewCapExtension: execution?.reviewCapExtension ?? 0,
|
|
377
505
|
};
|
|
378
506
|
}
|
|
379
507
|
|
|
@@ -515,6 +643,8 @@ function updatePanelWidget(ctx: ExtensionContext): void {
|
|
|
515
643
|
auditUndeterminable: current.audit.undeterminable,
|
|
516
644
|
findings: current.audit.findings,
|
|
517
645
|
reviewRunning: current.audit.running === true || current.review.inFlight !== null,
|
|
646
|
+
blockedTasks: blockedTaskIds(current),
|
|
647
|
+
blockedRound: current.blocked?.round ?? null,
|
|
518
648
|
startedAt: current.startedAt,
|
|
519
649
|
usage: current.usage,
|
|
520
650
|
});
|
|
@@ -540,6 +670,12 @@ export function updateStatusWidget(ctx: ExtensionContext): void {
|
|
|
540
670
|
auditUndeterminable: execution.audit.undeterminable,
|
|
541
671
|
findings: execution.audit.findings,
|
|
542
672
|
reviewRunning: execution.audit.running === true || execution.review.inFlight !== null,
|
|
673
|
+
blockedTasks: blockedTaskIds(execution),
|
|
674
|
+
blockedRound: execution.blocked?.round ?? null,
|
|
675
|
+
reviewBudget: execution.reviewBudget ?? null,
|
|
676
|
+
reviewRoundsTotal: execution.reviewRoundsTotal,
|
|
677
|
+
reviewCapExtension: execution.reviewCapExtension,
|
|
678
|
+
reviewBudgetDefaulted: execution.reviewBudgetDefaulted === true,
|
|
543
679
|
});
|
|
544
680
|
ctx.ui.setStatus("pi-plans", ctx.ui.theme.fg("accent", formatDashboardSummaryLine(model)));
|
|
545
681
|
return;
|
|
@@ -611,6 +747,15 @@ function persist(ctx: ExtensionContext): void {
|
|
|
611
747
|
startedAt: execution.startedAt,
|
|
612
748
|
usage: execution.usage,
|
|
613
749
|
stall: execution.stall,
|
|
750
|
+
blocked: execution.blocked,
|
|
751
|
+
// v0.9.3 (round-1 F-005): the review budget rides the session snapshot
|
|
752
|
+
// too — a /reload restore rebuilds the live state from it, and a
|
|
753
|
+
// restored counter must NOT hand the run a free unlimited window.
|
|
754
|
+
reviewBudget: execution.reviewBudget,
|
|
755
|
+
reviewBudgetDefaulted: execution.reviewBudgetDefaulted,
|
|
756
|
+
reviewRoundsTotal: execution.reviewRoundsTotal,
|
|
757
|
+
reviewCapExtension: execution.reviewCapExtension,
|
|
758
|
+
reviewNoProgress: execution.reviewNoProgress,
|
|
614
759
|
audit: { rounds: execution.audit.rounds, failed: execution.audit.failed, findings: execution.audit.findings },
|
|
615
760
|
});
|
|
616
761
|
}
|
|
@@ -654,9 +799,15 @@ export async function startExecution(
|
|
|
654
799
|
usage: { inToks: 0, outToks: 0 },
|
|
655
800
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
656
801
|
stall: { rounds: 0, lastSnapshot: null, paused: false },
|
|
802
|
+
blocked: null,
|
|
657
803
|
audit: { rounds: 0, failed: [], undeterminable: [], findings: [], running: false },
|
|
658
|
-
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
804
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
659
805
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
806
|
+
// v0.9.3: a fresh handoff starts with NO budget — the picker resolves it
|
|
807
|
+
// (or the no-UI default applies) right before round 1.
|
|
808
|
+
reviewBudget: undefined,
|
|
809
|
+
reviewRoundsTotal: 0,
|
|
810
|
+
reviewCapExtension: 0,
|
|
660
811
|
};
|
|
661
812
|
// Seed the watchdog baseline only after `execution` points at the new state
|
|
662
813
|
// (stallSnapshot reads the live execution).
|
|
@@ -716,11 +867,21 @@ export function persistTaskProgress(ctx: ExtensionContext): void {
|
|
|
716
867
|
// Any task-state change resets the stall watchdog baseline.
|
|
717
868
|
execution.stall.rounds = 0;
|
|
718
869
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
870
|
+
// v0.9.2: closing a reopened task shrinks the blocker; when the last one
|
|
871
|
+
// closes the record is dropped so no stale blocker outlives the repair.
|
|
872
|
+
if (execution.blocked) {
|
|
873
|
+
execution.blocked.tasks = blockedReviewTasks(execution.tasks, execution.blocked.rolledBack);
|
|
874
|
+
if (execution.blocked.tasks.length === 0) {
|
|
875
|
+
execution.blocked = null;
|
|
876
|
+
execution.blockedWakeTasks = undefined;
|
|
877
|
+
}
|
|
878
|
+
}
|
|
719
879
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
720
880
|
applyExecutionProgress(cp, {
|
|
721
881
|
tasks: taskProgressMap(execution!.tasks),
|
|
722
882
|
doneVcIds: execution!.items.filter((item) => item.done).map((item) => item.id),
|
|
723
883
|
stallRounds: execution!.stall.rounds,
|
|
884
|
+
blocked: execution!.blocked,
|
|
724
885
|
audit: {
|
|
725
886
|
rounds: execution!.audit.rounds,
|
|
726
887
|
lastResult: execution!.audit.failed.length > 0 ? execution!.audit.failed.join(",") : undefined,
|
|
@@ -884,14 +1045,102 @@ export function registerExecutionTurnHandlers(
|
|
|
884
1045
|
*
|
|
885
1046
|
* Completion stays fail-closed AND never fail-open: a run completes only
|
|
886
1047
|
* when every pending check was affirmatively passed. */
|
|
887
|
-
const REVIEW_CAP_PAUSE_PREFIX = "execution review exhausted";
|
|
888
|
-
/** v0.7 protocol value — dual-matched for one release so checkpoints written
|
|
889
|
-
* by older builds keep their cap pause recognized on restore. */
|
|
890
|
-
const LEGACY_AUDIT_CAP_PAUSE_PREFIX = "completion audit exhausted";
|
|
891
|
-
|
|
892
1048
|
function isReviewCapPause(reason: string | undefined | null): boolean {
|
|
893
|
-
|
|
894
|
-
|
|
1049
|
+
return isReviewPauseReason(reason);
|
|
1050
|
+
}
|
|
1051
|
+
|
|
1052
|
+
/** v0.9.3: the counters that bound an unlimited budget. */
|
|
1053
|
+
function budgetCounters(ex: ExecState): { reviewRoundsTotal: number; reviewCapExtension: number } {
|
|
1054
|
+
return { reviewRoundsTotal: ex.reviewRoundsTotal, reviewCapExtension: ex.reviewCapExtension };
|
|
1055
|
+
}
|
|
1056
|
+
|
|
1057
|
+
/** The budget in force for this run; a legacy checkpoint resolves to 5 during
|
|
1058
|
+
* load, so this is only a guard for hand-built test states. */
|
|
1059
|
+
function activeBudget(ex: ExecState): ReviewBudget {
|
|
1060
|
+
return ex.reviewBudget ?? LEGACY_REVIEW_MAX_ROUNDS;
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1063
|
+
function budgetSpent(ex: ExecState): boolean {
|
|
1064
|
+
return budgetExhausted(activeBudget(ex), ex.audit.rounds, budgetCounters(ex));
|
|
1065
|
+
}
|
|
1066
|
+
|
|
1067
|
+
/** `3` / `∞` — the budget as shown in status lines and messages. */
|
|
1068
|
+
function budgetLabel(ex: ExecState): string {
|
|
1069
|
+
return formatReviewBudget(activeBudget(ex));
|
|
1070
|
+
}
|
|
1071
|
+
|
|
1072
|
+
/**
|
|
1073
|
+
* v0.9.3: whether the budget panel may be shown in THIS session. Beyond the
|
|
1074
|
+
* native-surface check, auto-approve sessions never ask — the recorded plan
|
|
1075
|
+
* decision routes headless/auto-approve runs to the default budget
|
|
1076
|
+
* (`PI_PLANS_AUTO_APPROVE=1` exists for unattended harnesses, where a panel
|
|
1077
|
+
* would either hang or answer meaninglessly).
|
|
1078
|
+
*/
|
|
1079
|
+
function reviewBudgetPanelUsable(ctx: ExtensionContext): boolean {
|
|
1080
|
+
return !isAutoApproveEnabledLocal() && reviewBudgetPanelAvailable(ctx);
|
|
1081
|
+
}
|
|
1082
|
+
|
|
1083
|
+
/** Persist a freshly resolved budget (and the counters) in one revision. */
|
|
1084
|
+
function persistResolvedBudget(ctx: ExtensionContext, ex: ExecState): void {
|
|
1085
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
1086
|
+
applyExecutionProgress(cp, {
|
|
1087
|
+
reviewBudget: ex.reviewBudget ?? null,
|
|
1088
|
+
reviewBudgetDefaulted: ex.reviewBudgetDefaulted === true,
|
|
1089
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1090
|
+
reviewCapExtension: ex.reviewCapExtension,
|
|
1091
|
+
}),
|
|
1092
|
+
);
|
|
1093
|
+
persist(ctx);
|
|
1094
|
+
updateStatusWidget(ctx);
|
|
1095
|
+
}
|
|
1096
|
+
|
|
1097
|
+
/** One visible note whenever the fallback (rather than a user pick) decided
|
|
1098
|
+
* the budget — never a silent bound (v0.9.3, Q-3). */
|
|
1099
|
+
function notifyDefaultBudget(ex: ExecState, why: "no-panel" | "cancelled"): void {
|
|
1100
|
+
const budget = formatReviewBudget(ex.reviewBudget ?? DEFAULT_REVIEW_BUDGET);
|
|
1101
|
+
messaging().sendMessage(
|
|
1102
|
+
{
|
|
1103
|
+
customType: "pi-plans-review-budget-default",
|
|
1104
|
+
content:
|
|
1105
|
+
why === "no-panel"
|
|
1106
|
+
? `**pi-plans: execution review budget: ${budget} (default)** — no budget panel is available in this session, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`
|
|
1107
|
+
: `**pi-plans: execution review budget: ${budget} (default)** — the budget panel was closed without a choice, so the default budget applies. The review runs up to ${budget} round(s) before pausing for /plans-execute.`,
|
|
1108
|
+
display: true,
|
|
1109
|
+
},
|
|
1110
|
+
{ triggerTurn: false },
|
|
1111
|
+
);
|
|
1112
|
+
}
|
|
1113
|
+
|
|
1114
|
+
/** Synchronous fallback: apply (and persist) the default budget. Used when no
|
|
1115
|
+
* panel exists at all — the headless/print/json and auto-approve paths — so
|
|
1116
|
+
* the first round needs no extra async hop. */
|
|
1117
|
+
function applyDefaultReviewBudget(ctx: ExtensionContext, ex: ExecState): void {
|
|
1118
|
+
if (ex.reviewBudget !== undefined) return;
|
|
1119
|
+
ex.reviewBudget = DEFAULT_REVIEW_BUDGET;
|
|
1120
|
+
ex.reviewBudgetDefaulted = true;
|
|
1121
|
+
notifyDefaultBudget(ex, "no-panel");
|
|
1122
|
+
persistResolvedBudget(ctx, ex);
|
|
1123
|
+
}
|
|
1124
|
+
|
|
1125
|
+
/**
|
|
1126
|
+
* Resolve the per-run review budget exactly once, immediately before the first
|
|
1127
|
+
* round. The panel path is async; every other case is handled up front by
|
|
1128
|
+
* `applyDefaultReviewBudget`. `budgetAsking` keeps a second settle (or a
|
|
1129
|
+
* restore while the panel is open) from opening a second panel.
|
|
1130
|
+
*/
|
|
1131
|
+
async function askReviewBudgetForRun(ctx: ExtensionContext, ex: ExecState): Promise<void> {
|
|
1132
|
+
if (ex.reviewBudget !== undefined || ex.review.budgetAsking) return;
|
|
1133
|
+
ex.review.budgetAsking = true;
|
|
1134
|
+
try {
|
|
1135
|
+
const picked = await askReviewBudget(ctx, ex.uiLanguage, undefined);
|
|
1136
|
+
if (execution !== ex || ex.reviewBudget !== undefined) return;
|
|
1137
|
+
ex.reviewBudget = picked ?? DEFAULT_REVIEW_BUDGET;
|
|
1138
|
+
ex.reviewBudgetDefaulted = picked === null;
|
|
1139
|
+
if (picked === null) notifyDefaultBudget(ex, "cancelled");
|
|
1140
|
+
persistResolvedBudget(ctx, ex);
|
|
1141
|
+
} finally {
|
|
1142
|
+
ex.review.budgetAsking = false;
|
|
1143
|
+
}
|
|
895
1144
|
}
|
|
896
1145
|
|
|
897
1146
|
/** The currently-running (or self-scheduling) review chain; the sanctioned
|
|
@@ -933,8 +1182,56 @@ function toReviewFindings(records?: ReviewFindingRecord[]): ReviewFinding[] {
|
|
|
933
1182
|
* the review owed even when every check is done (the stranded-high path),
|
|
934
1183
|
* mirroring how a failed check keeps it owed today. */
|
|
935
1184
|
function reviewOwed(ex: ExecState): boolean {
|
|
936
|
-
return allTasksTerminal(ex.tasks)
|
|
937
|
-
|
|
1185
|
+
return allTasksTerminal(ex.tasks) && reviewOutstanding(ex);
|
|
1186
|
+
}
|
|
1187
|
+
|
|
1188
|
+
/** v0.9.2: the review is outstanding wherever the tree stands — checks still
|
|
1189
|
+
* owed or an unresolved high finding. `reviewOwed` ANDs this with
|
|
1190
|
+
* allTasksTerminal; the blocked wake needs exactly the half that stays true
|
|
1191
|
+
* while tasks are open, because that is the state in which the review cannot
|
|
1192
|
+
* start (and in which `pendingAudit` is false by construction). */
|
|
1193
|
+
function reviewOutstanding(ex: ExecState): boolean {
|
|
1194
|
+
return auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0;
|
|
1195
|
+
}
|
|
1196
|
+
|
|
1197
|
+
/** v0.9.2: live blocker ids — the newest failed round's rollback set
|
|
1198
|
+
* intersected with the still-open tasks. Always recomputed from the tree: the
|
|
1199
|
+
* stored `tasks` list may predate the last status change. */
|
|
1200
|
+
function blockedTaskIds(ex: ExecState): string[] {
|
|
1201
|
+
return ex.blocked ? blockedReviewTasks(ex.tasks, ex.blocked.rolledBack) : [];
|
|
1202
|
+
}
|
|
1203
|
+
|
|
1204
|
+
/** v0.9.2: one visible system line per escalation level, so the user sees the
|
|
1205
|
+
* loop escalating before the watchdog pauses it. Memory-only latch (the
|
|
1206
|
+
* restored state re-notifies once, which is the useful behavior). */
|
|
1207
|
+
function notifyBlockedEscalation(ctx: ExtensionContext, ex: ExecState): void {
|
|
1208
|
+
const level = ex.blocked?.escalatedRounds ?? 0;
|
|
1209
|
+
if (!ex.blocked || level === 0 || ex.blockedNotifiedLevel === level) return;
|
|
1210
|
+
ex.blockedNotifiedLevel = level;
|
|
1211
|
+
const ids = blockedTaskIds(ex);
|
|
1212
|
+
try {
|
|
1213
|
+
messaging().sendMessage(
|
|
1214
|
+
{
|
|
1215
|
+
customType: EXECUTION_BLOCKED_CUSTOM_TYPE,
|
|
1216
|
+
content: `**pi-plans: execution blocked (wake ${level}/${STALL_MAX_ROUNDS})** — review round ${ex.blocked.round + 1} cannot start: ${ids.join(", ")} ${ids.length === 1 ? "was" : "were"} reopened by round ${ex.blocked.round} and ${ids.length === 1 ? "is" : "are"} still open. Close them with \`plans_update_task\` (the review starts by itself once every task is terminal).`,
|
|
1217
|
+
display: true,
|
|
1218
|
+
},
|
|
1219
|
+
{ triggerTurn: false },
|
|
1220
|
+
);
|
|
1221
|
+
} catch {
|
|
1222
|
+
/* best-effort: the wake below still carries the same blocker */
|
|
1223
|
+
}
|
|
1224
|
+
}
|
|
1225
|
+
|
|
1226
|
+
/** v0.9.2: the pause reason names the real blocker instead of the watchdog's
|
|
1227
|
+
* own metric. Single line, never a review-cap prefix (`isReviewCapPause`
|
|
1228
|
+
* matches "execution review exhausted" / "completion audit exhausted"). */
|
|
1229
|
+
function blockedPauseReason(ex: ExecState): string {
|
|
1230
|
+
const ids = blockedTaskIds(ex);
|
|
1231
|
+
const round = ex.blocked?.round ?? ex.audit.rounds;
|
|
1232
|
+
const head = ids.slice(0, 3).join(", ");
|
|
1233
|
+
const more = ids.length > 3 ? `, +${ids.length - 3} more` : "";
|
|
1234
|
+
return `blocked: review round ${round + 1} cannot start — ${ids.length} task(s) reopened by round ${round} still open (${head}${more})`;
|
|
938
1235
|
}
|
|
939
1236
|
|
|
940
1237
|
function runDirOf(ctx: ExtensionContext): string | null {
|
|
@@ -1076,7 +1373,13 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
|
1076
1373
|
: highs.length > 0
|
|
1077
1374
|
? `high findings: ${highs.map((f) => f.id).join(", ")}`
|
|
1078
1375
|
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1079
|
-
|
|
1376
|
+
// v0.9.3: the reason names the budget actually in force — the numeric round
|
|
1377
|
+
// count, or the unlimited hard cap that stopped the loop.
|
|
1378
|
+
const scope =
|
|
1379
|
+
activeBudget(ex) === "unlimited"
|
|
1380
|
+
? `${unlimitedHardCapCeiling(budgetCounters(ex))} rounds (unlimited budget safety cap)`
|
|
1381
|
+
: `${activeBudget(ex)} round${activeBudget(ex) === 1 ? "" : "s"}`;
|
|
1382
|
+
const reason = `${REVIEW_CAP_PAUSE_PREFIX} ${scope} (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the ${formatReviewBudget(activeBudget(ex))} picker; ordinary messages and session restores do not). (Or close the failed checks' tasks as skipped to pass them as skipped-pass.)`;
|
|
1080
1383
|
pauseForStall(ctx, reason);
|
|
1081
1384
|
// In-band, headless-visible pause signal (the pi-plans-exec-stop pattern):
|
|
1082
1385
|
// pauseForStall's ui.notify is optional and absent headless, so the pause
|
|
@@ -1087,6 +1390,41 @@ function pauseReviewCap(ctx: ExtensionContext, ex: ExecState): void {
|
|
|
1087
1390
|
);
|
|
1088
1391
|
}
|
|
1089
1392
|
|
|
1393
|
+
/** v0.9.3: the unlimited budget's no-progress valve — three consecutive
|
|
1394
|
+
* committed rounds with an identical outcome signature cannot converge, so
|
|
1395
|
+
* the run pauses fail-closed instead of burning rounds silently. */
|
|
1396
|
+
function pauseReviewNoProgress(ctx: ExtensionContext, ex: ExecState): void {
|
|
1397
|
+
const streak = ex.reviewNoProgress?.streak ?? NO_PROGRESS_MAX_STREAK;
|
|
1398
|
+
const highs = unresolvedHighFindings(ex);
|
|
1399
|
+
const detail =
|
|
1400
|
+
ex.audit.failed.length > 0
|
|
1401
|
+
? `failed: ${ex.audit.failed.join(", ")}${highs.length > 0 ? `; high findings: ${highs.map((f) => f.id).join(", ")}` : ""}`
|
|
1402
|
+
: highs.length > 0
|
|
1403
|
+
? `high findings: ${highs.map((f) => f.id).join(", ")}`
|
|
1404
|
+
: `unreadable verdicts: ${ex.audit.undeterminable.join(", ") || "unknown"}`;
|
|
1405
|
+
const reason = `${REVIEW_NO_PROGRESS_PAUSE_PREFIX} — ${streak} consecutive rounds reported the same outcome (${detail}). Only /plans-execute — an explicit user confirmation — grants a fresh budget (and re-opens the budget picker; ordinary messages and session restores do not).`;
|
|
1406
|
+
pauseForStall(ctx, reason);
|
|
1407
|
+
messaging().sendMessage(
|
|
1408
|
+
{ customType: "pi-plans-review-paused", content: `**pi-plans: ${reason}**`, display: true },
|
|
1409
|
+
{ triggerTurn: false },
|
|
1410
|
+
);
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
/** v0.9.3: the single gate every round spawn passes through. Returns false
|
|
1414
|
+
* when the run was paused (no round may start). The unlimited valve order is
|
|
1415
|
+
* deliberate: no-progress is checked first so its reason wins when both hold. */
|
|
1416
|
+
function reviewBudgetGate(ctx: ExtensionContext, ex: ExecState): boolean {
|
|
1417
|
+
if (activeBudget(ex) === "unlimited" && noProgressTripped(ex.reviewNoProgress)) {
|
|
1418
|
+
pauseReviewNoProgress(ctx, ex);
|
|
1419
|
+
return false;
|
|
1420
|
+
}
|
|
1421
|
+
if (budgetSpent(ex)) {
|
|
1422
|
+
pauseReviewCap(ctx, ex);
|
|
1423
|
+
return false;
|
|
1424
|
+
}
|
|
1425
|
+
return true;
|
|
1426
|
+
}
|
|
1427
|
+
|
|
1090
1428
|
async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
1091
1429
|
if (!execution) return;
|
|
1092
1430
|
const ex = execution;
|
|
@@ -1105,11 +1443,27 @@ async function startReviewRound(ctx: ExtensionContext): Promise<void> {
|
|
|
1105
1443
|
await completeExecution(ctx);
|
|
1106
1444
|
return;
|
|
1107
1445
|
}
|
|
1108
|
-
if (ex.stall.paused || ex.review.inFlight) return;
|
|
1109
|
-
|
|
1110
|
-
|
|
1111
|
-
|
|
1446
|
+
if (ex.stall.paused || ex.review.inFlight || ex.review.budgetAsking) return;
|
|
1447
|
+
// v0.9.3: the budget is resolved right here — every task is terminal, the
|
|
1448
|
+
// review is genuinely owed, and this is the last moment before round 1.
|
|
1449
|
+
// No panel → the default applies synchronously (no extra tick, so a
|
|
1450
|
+
// detached settle still shows the round in flight immediately); a panel →
|
|
1451
|
+
// one await, guarded against a second settle by `budgetAsking`.
|
|
1452
|
+
if (ex.reviewBudget === undefined && !reviewBudgetPanelUsable(ctx)) applyDefaultReviewBudget(ctx, ex);
|
|
1453
|
+
if (ex.reviewBudget === undefined) {
|
|
1454
|
+
await askReviewBudgetForRun(ctx, ex);
|
|
1455
|
+
if (execution !== ex || ex.reviewBudget === undefined) return;
|
|
1456
|
+
}
|
|
1457
|
+
if (pendingChecks.length === 0) {
|
|
1458
|
+
// Every verification check is satisfied; only unresolved highs keep the
|
|
1459
|
+
// review owed. With the budget spent there is no round left to buy, so
|
|
1460
|
+
// the run completes and DISCLOSES the highs (v0.9.3, Q-2).
|
|
1461
|
+
if (budgetSpent(ex)) {
|
|
1462
|
+
await completeExecution(ctx);
|
|
1463
|
+
return;
|
|
1464
|
+
}
|
|
1112
1465
|
}
|
|
1466
|
+
if (!reviewBudgetGate(ctx, ex)) return;
|
|
1113
1467
|
// Phase transition: the executor is done with its tasks; the review loop
|
|
1114
1468
|
// owns the run until it converges (or pauses at the cap).
|
|
1115
1469
|
setRunStatusForReview(ctx, "verifying");
|
|
@@ -1320,6 +1674,9 @@ async function commitReviewOutcome(
|
|
|
1320
1674
|
// The budget is charged only when an outcome commits — never on discard
|
|
1321
1675
|
// or cancellation (CF2-003).
|
|
1322
1676
|
ex.audit.rounds = round.budgetRound;
|
|
1677
|
+
// v0.9.3: the run-cumulative counter never resets — it is what bounds an
|
|
1678
|
+
// unlimited budget across grants (Q-4).
|
|
1679
|
+
ex.reviewRoundsTotal += 1;
|
|
1323
1680
|
const passed = pendingIds.filter((id) => (outcome?.passed ?? []).includes(id));
|
|
1324
1681
|
const failed = pendingIds.filter((id) => (outcome?.failed ?? []).includes(id));
|
|
1325
1682
|
// Anything the round neither passed nor failed is undeterminable: the
|
|
@@ -1337,6 +1694,12 @@ async function commitReviewOutcome(
|
|
|
1337
1694
|
const findings = reported ? outcome.findings : ex.audit.findings;
|
|
1338
1695
|
ex.audit.findings = findings;
|
|
1339
1696
|
const highs = findings.filter((f) => f.severity === "high");
|
|
1697
|
+
// v0.9.3: the no-progress valve's signature is computed HERE, from the
|
|
1698
|
+
// post-classification triple — a spawn-failure round (`outcome === null`)
|
|
1699
|
+
// and a discard synthesis have no findings array of their own, but the
|
|
1700
|
+
// preserved unresolved set plus the derived verdicts still describe a
|
|
1701
|
+
// concrete, comparable outcome (round-1 F-002).
|
|
1702
|
+
ex.reviewNoProgress = bumpNoProgress(ex.reviewNoProgress, noProgressSignature(failed, undeterminable, highs.map((f) => f.id)));
|
|
1340
1703
|
// Findings are actionable only when this round actually reported them;
|
|
1341
1704
|
// see the fix-loop branch below (v0.9.1, F-001).
|
|
1342
1705
|
const actionableHighs = reported ? highs : [];
|
|
@@ -1371,6 +1734,40 @@ async function commitReviewOutcome(
|
|
|
1371
1734
|
if (item) item.done = true;
|
|
1372
1735
|
}
|
|
1373
1736
|
|
|
1737
|
+
// v0.9.3 (Q-2/F-003): the exhausted-budget completion. EVERY check this
|
|
1738
|
+
// round still owed affirmed, only unresolved highs remain, and the budget
|
|
1739
|
+
// cannot buy another round — the run completes, tolerating the highs (they
|
|
1740
|
+
// stay in the round report and the checkpoint; `completeExecution`
|
|
1741
|
+
// discloses them). Evaluated BEFORE the fix branch on purpose: the
|
|
1742
|
+
// tolerated round must not roll back tasks, append plan tasks, rewrite the
|
|
1743
|
+
// approved plan, or wake the executor for a run that is about to be done.
|
|
1744
|
+
if (budgetSpent(ex) && passed.length === pendingIds.length && highs.length > 0) {
|
|
1745
|
+
ex.audit.failed = [];
|
|
1746
|
+
ex.audit.undeterminable = [];
|
|
1747
|
+
ex.blocked = null;
|
|
1748
|
+
withExecutionCheckpoint(ctx, (cp) =>
|
|
1749
|
+
applyExecutionProgress(cp, {
|
|
1750
|
+
tasks: taskProgressMap(ex.tasks),
|
|
1751
|
+
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1752
|
+
blocked: null,
|
|
1753
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1754
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1755
|
+
audit: { rounds: ex.audit.rounds, passed: true, findings },
|
|
1756
|
+
}),
|
|
1757
|
+
);
|
|
1758
|
+
persist(ctx);
|
|
1759
|
+
messaging().sendMessage(
|
|
1760
|
+
{
|
|
1761
|
+
customType: "pi-plans-review-tolerated",
|
|
1762
|
+
content: `**pi-plans: review budget exhausted (${budgetLabel(ex)}) — completing with ${highs.length} unresolved high finding(s): ${highs.map((f) => f.id).join(", ")}** — every verification check passed; the finding(s) remain in the round reports.`,
|
|
1763
|
+
display: true,
|
|
1764
|
+
},
|
|
1765
|
+
{ triggerTurn: false },
|
|
1766
|
+
);
|
|
1767
|
+
await completeExecution(ctx);
|
|
1768
|
+
return;
|
|
1769
|
+
}
|
|
1770
|
+
|
|
1374
1771
|
// v0.9 fix loop — evaluated BEFORE the completion branch so an unresolved
|
|
1375
1772
|
// high finding can never complete the run (liveness). Failed checks and
|
|
1376
1773
|
// high findings drive ONE union rollback and exactly one executor wake;
|
|
@@ -1398,6 +1795,20 @@ async function commitReviewOutcome(
|
|
|
1398
1795
|
const unmappedHighs = actionableHighs.filter((h) => !h.taskIds.some((id) => knownIds.has(id)));
|
|
1399
1796
|
const amended = unmappedHighs.length > 0 ? appendFindingTasks(ex, unmappedHighs) : [];
|
|
1400
1797
|
const allRolledBack = [...new Set([...rolledBack, ...highRolledBack])];
|
|
1798
|
+
// v0.9.2: capture the round's rollback set HERE. The reopen helpers above
|
|
1799
|
+
// mutate the tree and report only flipped nodes, so this is the only
|
|
1800
|
+
// moment the authoritative provenance exists; the still-open subset is
|
|
1801
|
+
// what blocks the next round and what wakes/pauses name explicitly.
|
|
1802
|
+
ex.blocked = allRolledBack.length > 0
|
|
1803
|
+
? {
|
|
1804
|
+
rolledBack: [...allRolledBack],
|
|
1805
|
+
tasks: blockedReviewTasks(ex.tasks, allRolledBack),
|
|
1806
|
+
round: ex.audit.rounds,
|
|
1807
|
+
escalatedRounds: 0,
|
|
1808
|
+
since: utcNow(),
|
|
1809
|
+
}
|
|
1810
|
+
: null;
|
|
1811
|
+
ex.blockedWakeTasks = undefined;
|
|
1401
1812
|
withExecutionCheckpoint(ctx, (cp) => {
|
|
1402
1813
|
// v0.9.1 (F-002): appending finding tasks rewrote the approved plan;
|
|
1403
1814
|
// re-stamp the checkpoint's plan identity in the same revision so a
|
|
@@ -1409,6 +1820,9 @@ async function commitReviewOutcome(
|
|
|
1409
1820
|
return applyExecutionProgress(amendedCp, {
|
|
1410
1821
|
tasks: taskProgressMap(ex.tasks),
|
|
1411
1822
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1823
|
+
blocked: ex.blocked,
|
|
1824
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1825
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1412
1826
|
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || `highs: ${actionableHighs.map((h) => h.id).join(",")}`, findings },
|
|
1413
1827
|
});
|
|
1414
1828
|
});
|
|
@@ -1442,7 +1856,7 @@ async function commitReviewOutcome(
|
|
|
1442
1856
|
? `**pi-plans: execution review round ${ex.audit.rounds} found ${actionableHighs.length} high-severity finding(s)**${failed.length > 0 ? ` and failed checks: ${failed.join(", ")}` : ""}.`
|
|
1443
1857
|
: `**pi-plans: execution review round ${ex.audit.rounds} failed** — checks: ${failed.join(", ")}.`;
|
|
1444
1858
|
const findingsBlock = actionableHighs.length > 0 ? `\n\nHigh findings:\n${highLines}` : "";
|
|
1445
|
-
const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${ex
|
|
1859
|
+
const content = `${findingsLead} Rolled back tasks: ${allRolledBack.join(", ") || "(none covered)"}${amended.length > 0 ? `. Tasks appended to the plan for unmapped findings: ${amended.join(", ")}` : ""}.${findingsBlock}\n\nFix them and re-close the affected tasks with \`plans_update_task\`; the review reruns automatically once all tasks are terminal again.${budgetSpent(ex) ? (activeBudget(ex) === "unlimited" ? ` This was round ${ex.reviewRoundsTotal} against the unlimited budget's ${unlimitedHardCapCeiling(budgetCounters(ex))}-round safety cap: the next terminal-task cycle pauses the run for review.` : ` This was round ${ex.audit.rounds} of ${budgetLabel(ex)}: the next terminal-task cycle pauses the run for review (or completes if every check passed).`) : ""}${stranded ? ` No task covers the finding(s) and none could be appended — the task tree stayed terminal; the next settle re-runs the review automatically.` : ""}\n\n${reportRef}`;
|
|
1446
1860
|
messaging().sendMessage(
|
|
1447
1861
|
{
|
|
1448
1862
|
customType: "pi-plans-audit-failed",
|
|
@@ -1465,10 +1879,15 @@ async function commitReviewOutcome(
|
|
|
1465
1879
|
if (passed.length === pendingIds.length && highs.length === 0) {
|
|
1466
1880
|
ex.audit.failed = [];
|
|
1467
1881
|
ex.audit.undeterminable = [];
|
|
1882
|
+
// v0.9.2: a passing audit ends the blocker — the review consumed it.
|
|
1883
|
+
ex.blocked = null;
|
|
1468
1884
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
1469
1885
|
applyExecutionProgress(cp, {
|
|
1470
1886
|
tasks: taskProgressMap(ex.tasks),
|
|
1471
1887
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1888
|
+
blocked: null,
|
|
1889
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1890
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1472
1891
|
audit: { rounds: ex.audit.rounds, passed: true, findings },
|
|
1473
1892
|
}),
|
|
1474
1893
|
);
|
|
@@ -1484,6 +1903,8 @@ async function commitReviewOutcome(
|
|
|
1484
1903
|
applyExecutionProgress(cp, {
|
|
1485
1904
|
tasks: taskProgressMap(ex.tasks),
|
|
1486
1905
|
doneVcIds: ex.items.filter((item) => item.done).map((item) => item.id),
|
|
1906
|
+
reviewRoundsTotal: ex.reviewRoundsTotal,
|
|
1907
|
+
reviewNoProgress: ex.reviewNoProgress ?? null,
|
|
1487
1908
|
audit: { rounds: ex.audit.rounds, lastResult: failed.join(",") || undefined, findings },
|
|
1488
1909
|
}),
|
|
1489
1910
|
);
|
|
@@ -1496,10 +1917,11 @@ async function maybeContinueReview(ctx: ExtensionContext, ex: ExecState): Promis
|
|
|
1496
1917
|
if (execution !== ex) return;
|
|
1497
1918
|
if (!reviewOwed(ex)) return;
|
|
1498
1919
|
if (ex.stall.paused) return;
|
|
1499
|
-
|
|
1500
|
-
|
|
1501
|
-
|
|
1502
|
-
|
|
1920
|
+
// The budget gate owns every pause (numeric exhaustion, the unlimited hard
|
|
1921
|
+
// cap, and the unlimited no-progress valve). An exhausted budget with every
|
|
1922
|
+
// check satisfied has already completed inside commitReviewOutcome; here
|
|
1923
|
+
// the review is still owed, so a spent budget pauses.
|
|
1924
|
+
if (!reviewBudgetGate(ctx, ex)) return;
|
|
1503
1925
|
await startReviewRound(ctx);
|
|
1504
1926
|
}
|
|
1505
1927
|
|
|
@@ -2085,6 +2507,8 @@ function pendingAudit(ex: ExecState | null = execution): ex is ExecState {
|
|
|
2085
2507
|
// This also bounds the zero-input continue loop in agent_before_settle.
|
|
2086
2508
|
if (ex.stall.paused) return false;
|
|
2087
2509
|
if (ex.review.inFlight) return false;
|
|
2510
|
+
// The budget panel is open: the review is being resolved, not owed anew.
|
|
2511
|
+
if (ex.review.budgetAsking) return false;
|
|
2088
2512
|
return allTasksTerminal(ex.tasks)
|
|
2089
2513
|
&& (auditableChecks(ex.items, ex.tasks).some((item) => !item.done) || unresolvedHighFindings(ex).length > 0);
|
|
2090
2514
|
}
|
|
@@ -2171,6 +2595,33 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
|
|
|
2171
2595
|
if (runtime.stopReason !== "stop" || !canWakeExecution(ctx, runtime)) return;
|
|
2172
2596
|
runtime.handled = true;
|
|
2173
2597
|
const ex = runtime.owner;
|
|
2598
|
+
// v0.9.2: while a failed round's tasks are still open, the next round
|
|
2599
|
+
// cannot start — and this ladder must NOT ride `stall.rounds`, which any
|
|
2600
|
+
// successful tool call resets: an executor that investigates but never
|
|
2601
|
+
// closes the reopened tasks has to escalate (and eventually pause) anyway.
|
|
2602
|
+
const blockedIds = blockedTaskIds(ex);
|
|
2603
|
+
if (ex.blocked && blockedIds.length > 0) {
|
|
2604
|
+
// Progress = fewer blockers than at the PREVIOUS blocked wake. The
|
|
2605
|
+
// stored `blocked.tasks` cannot serve as that baseline: every task
|
|
2606
|
+
// close re-syncs it, so it always equals the live set and the ladder
|
|
2607
|
+
// would only ever increment (review round 1, F-001). A same-size but
|
|
2608
|
+
// different set counts as no progress; a new member counts as none.
|
|
2609
|
+
const baseline = ex.blockedWakeTasks;
|
|
2610
|
+
const progressed = baseline !== undefined && blockedIds.length < baseline.length;
|
|
2611
|
+
ex.blockedWakeTasks = [...blockedIds];
|
|
2612
|
+
ex.blocked.escalatedRounds = progressed ? 0 : ex.blocked.escalatedRounds + 1;
|
|
2613
|
+
ex.blocked.tasks = blockedIds;
|
|
2614
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { blocked: ex.blocked }));
|
|
2615
|
+
if (ex.blocked.escalatedRounds >= STALL_MAX_ROUNDS) {
|
|
2616
|
+
pauseForStall(ctx, blockedPauseReason(ex));
|
|
2617
|
+
return;
|
|
2618
|
+
}
|
|
2619
|
+
notifyBlockedEscalation(ctx, ex);
|
|
2620
|
+
persist(ctx);
|
|
2621
|
+
updateStatusWidget(ctx);
|
|
2622
|
+
sendContinuationWake(ctx, runtime);
|
|
2623
|
+
return;
|
|
2624
|
+
}
|
|
2174
2625
|
const snapshot = stallSnapshot();
|
|
2175
2626
|
const changed = ex.stall.lastSnapshot !== null && snapshot !== ex.stall.lastSnapshot;
|
|
2176
2627
|
ex.stall.lastSnapshot = snapshot;
|
|
@@ -2190,8 +2641,10 @@ function maybeContinuationFollowUp(ctx: ExtensionContext): void {
|
|
|
2190
2641
|
|
|
2191
2642
|
export function filterGoalWaitMessages<T extends { customType?: string; details?: unknown }>(messages: T[]): T[] {
|
|
2192
2643
|
// v0.6.1: continuation wakes are one-shot; stale ones (including the
|
|
2193
|
-
// legacy v0.6.0 goal-wait type) never replay after a restart.
|
|
2644
|
+
// legacy v0.6.0 goal-wait type) never replay after a restart. v0.9.2: the
|
|
2645
|
+
// visible escalated-blocked line is one-shot for the same reason.
|
|
2194
2646
|
return messages.filter((message) => message.customType !== EXECUTION_CONTINUE_CUSTOM_TYPE
|
|
2647
|
+
&& message.customType !== EXECUTION_BLOCKED_CUSTOM_TYPE
|
|
2195
2648
|
&& message.customType !== LEGACY_GOAL_WAIT_CUSTOM_TYPE);
|
|
2196
2649
|
}
|
|
2197
2650
|
|
|
@@ -2200,18 +2653,20 @@ export function filterContinuationMessages<T extends { customType?: string; deta
|
|
|
2200
2653
|
}
|
|
2201
2654
|
|
|
2202
2655
|
/** Called for genuine user input or an explicit same-execution resume.
|
|
2203
|
-
* v0.8: a REVIEW
|
|
2204
|
-
* refill the
|
|
2656
|
+
* v0.8: a REVIEW pause is never lifted here — ordinary input must not
|
|
2657
|
+
* refill the review budget (CF2-004); only /plans-execute
|
|
2205
2658
|
* (resumeActiveExecution) is the explicit confirmation surface. Genuine
|
|
2206
2659
|
* stall pauses still clear on input as before. */
|
|
2207
2660
|
export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
2208
2661
|
const ex = getExecution();
|
|
2209
2662
|
if (!ex?.stall.paused || !currentContinuationRuntime(ctx)) return false;
|
|
2210
|
-
if (
|
|
2663
|
+
if (isReviewPauseReason(ex.stall.pausedReason)) {
|
|
2211
2664
|
// Surfaced once per input so the user is not left guessing why the run
|
|
2212
2665
|
// stays paused; the pause itself and the budget survive untouched.
|
|
2666
|
+
// v0.9.3: the note names the budget actually in force and the fact that
|
|
2667
|
+
// the confirmation re-opens the picker.
|
|
2213
2668
|
ctx.ui.notify?.(
|
|
2214
|
-
|
|
2669
|
+
`pi-plans: the review budget is exhausted (${budgetLabel(ex)}) — run /plans-execute to grant a fresh budget (it re-opens the round-count picker; that confirmation is the only surface that does).`,
|
|
2215
2670
|
"warning",
|
|
2216
2671
|
);
|
|
2217
2672
|
return false;
|
|
@@ -2220,18 +2675,61 @@ export function resumeGoalWaitIfPaused(ctx: ExtensionContext): boolean {
|
|
|
2220
2675
|
ex.stall.pausedReason = undefined;
|
|
2221
2676
|
ex.stall.rounds = 0;
|
|
2222
2677
|
ex.stall.lastSnapshot = stallSnapshot();
|
|
2223
|
-
|
|
2678
|
+
// v0.9.2: a resume grants a fresh escalation ladder (the blocker itself is
|
|
2679
|
+
// still recorded, so the next wake names it) and never a fresh review
|
|
2680
|
+
// budget — that stays `/plans-execute`-only.
|
|
2681
|
+
if (ex.blocked) {
|
|
2682
|
+
ex.blocked.escalatedRounds = 0;
|
|
2683
|
+
ex.blockedWakeTasks = undefined;
|
|
2684
|
+
}
|
|
2685
|
+
withExecutionCheckpoint(ctx, (cp) => applyExecutionProgress(cp, { pausedReason: null, blocked: ex.blocked }));
|
|
2224
2686
|
persist(ctx);
|
|
2225
2687
|
updateStatusWidget(ctx);
|
|
2226
2688
|
return true;
|
|
2227
2689
|
}
|
|
2228
2690
|
|
|
2229
|
-
|
|
2230
|
-
|
|
2231
|
-
|
|
2232
|
-
|
|
2691
|
+
/** Outcome of an explicit `/plans-execute` resume, so the calling tool can
|
|
2692
|
+
* report what actually happened (v0.9.3: the grant may re-open the picker). */
|
|
2693
|
+
export interface ResumeOutcome {
|
|
2694
|
+
resumed: boolean;
|
|
2695
|
+
/** Set when the pause was lifted by a budget grant. */
|
|
2696
|
+
grantedBudget?: ReviewBudget;
|
|
2697
|
+
/** True when the user closed the budget panel — the pause stands. */
|
|
2698
|
+
budgetDeclined?: boolean;
|
|
2699
|
+
}
|
|
2700
|
+
|
|
2701
|
+
export async function resumeActiveExecution(ctx: ExtensionContext): Promise<ResumeOutcome> {
|
|
2702
|
+
// v0.8: /plans-execute is THE explicit confirmation surface for a review
|
|
2703
|
+
// pause — the only place a fresh budget is granted (Q-confirm-surface).
|
|
2704
|
+
// Ordinary input and session restores never refill.
|
|
2705
|
+
// v0.9.3: the grant re-opens the budget picker with the current value
|
|
2706
|
+
// preselected; Esc keeps the run paused (an explicit confirmation is the
|
|
2707
|
+
// only way forward). Headless sessions, which have no panel to show, keep
|
|
2708
|
+
// the current budget instead of stranding the run.
|
|
2233
2709
|
const pausedEx = getExecution();
|
|
2234
|
-
if (pausedEx?.stall.paused &&
|
|
2710
|
+
if (pausedEx?.stall.paused && isReviewPauseReason(pausedEx.stall.pausedReason)) {
|
|
2711
|
+
const previous = activeBudget(pausedEx);
|
|
2712
|
+
let granted = previous;
|
|
2713
|
+
if (reviewBudgetPanelUsable(ctx)) {
|
|
2714
|
+
const picked = await askReviewBudget(ctx, pausedEx.uiLanguage, previous);
|
|
2715
|
+
if (picked === null) {
|
|
2716
|
+
messaging().sendMessage(
|
|
2717
|
+
{
|
|
2718
|
+
customType: "pi-plans-review-budget-declined",
|
|
2719
|
+
content: `**pi-plans: review budget unchanged (${formatReviewBudget(previous)})** — the run stays paused. Run /plans-execute and pick a round count to continue.`,
|
|
2720
|
+
display: true,
|
|
2721
|
+
},
|
|
2722
|
+
{ triggerTurn: false },
|
|
2723
|
+
);
|
|
2724
|
+
return { resumed: false, budgetDeclined: true };
|
|
2725
|
+
}
|
|
2726
|
+
granted = picked;
|
|
2727
|
+
// Q-4: only a grant that lands on `unlimited` lifts the hard cap —
|
|
2728
|
+
// the cumulative counter itself never resets.
|
|
2729
|
+
if (granted === "unlimited") {
|
|
2730
|
+
pausedEx.reviewCapExtension += UNLIMITED_HARD_CAP;
|
|
2731
|
+
}
|
|
2732
|
+
}
|
|
2235
2733
|
pausedEx.stall.paused = false;
|
|
2236
2734
|
pausedEx.stall.pausedReason = undefined;
|
|
2237
2735
|
pausedEx.stall.rounds = 0;
|
|
@@ -2239,28 +2737,45 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
|
2239
2737
|
pausedEx.audit.rounds = 0;
|
|
2240
2738
|
pausedEx.audit.failed = [];
|
|
2241
2739
|
pausedEx.audit.undeterminable = [];
|
|
2740
|
+
// v0.9.3: a fresh budget window also resets the no-progress valve.
|
|
2741
|
+
pausedEx.reviewNoProgress = undefined;
|
|
2742
|
+
// v0.9.2: the explicit confirmation also refreshes the blocked ladder
|
|
2743
|
+
// (the blocker set itself survives — it still names what stays open).
|
|
2744
|
+
if (pausedEx.blocked) {
|
|
2745
|
+
pausedEx.blocked.escalatedRounds = 0;
|
|
2746
|
+
pausedEx.blockedWakeTasks = undefined;
|
|
2747
|
+
}
|
|
2242
2748
|
// v0.9: the fresh budget inherits unresolved findings (stable ids keep
|
|
2243
2749
|
// counting) — only the round counter resets.
|
|
2244
2750
|
withExecutionCheckpoint(ctx, (cp) =>
|
|
2245
2751
|
applyExecutionProgress(cp, {
|
|
2246
2752
|
tasks: taskProgressMap(pausedEx.tasks),
|
|
2247
2753
|
audit: { rounds: 0, lastResult: undefined, findings: pausedEx.audit.findings },
|
|
2754
|
+
blocked: pausedEx.blocked,
|
|
2755
|
+
reviewBudget: granted,
|
|
2756
|
+
reviewBudgetDefaulted: pausedEx.reviewBudgetDefaulted === true,
|
|
2757
|
+
reviewRoundsTotal: pausedEx.reviewRoundsTotal,
|
|
2758
|
+
reviewCapExtension: pausedEx.reviewCapExtension,
|
|
2759
|
+
reviewNoProgress: null,
|
|
2248
2760
|
pausedReason: null,
|
|
2249
2761
|
}),
|
|
2250
2762
|
);
|
|
2251
2763
|
persist(ctx);
|
|
2252
2764
|
updateStatusWidget(ctx);
|
|
2765
|
+
const grantedNote = granted === "unlimited"
|
|
2766
|
+
? `unlimited review budget granted (hard cap now ${unlimitedHardCapCeiling(budgetCounters(pausedEx))} rounds; the run has spent ${pausedEx.reviewRoundsTotal})`
|
|
2767
|
+
: `fresh ${formatReviewBudget(granted)}-round review budget granted`;
|
|
2253
2768
|
messaging().sendMessage(
|
|
2254
2769
|
{
|
|
2255
2770
|
customType: "pi-plans-review-budget-granted",
|
|
2256
|
-
content:
|
|
2771
|
+
content: `**pi-plans: ${grantedNote}** — the execution review resumes now.`,
|
|
2257
2772
|
display: true,
|
|
2258
2773
|
},
|
|
2259
2774
|
{ triggerTurn: false },
|
|
2260
2775
|
);
|
|
2261
2776
|
const grantChain = launchReviewRound(ctx);
|
|
2262
2777
|
if (grantChain) void grantChain.catch(() => { /* surfaced via the review messages */ });
|
|
2263
|
-
return true;
|
|
2778
|
+
return { resumed: true, grantedBudget: granted };
|
|
2264
2779
|
}
|
|
2265
2780
|
// v0.7.1 (root cause A): a terminal-but-unaudited run used to fall through
|
|
2266
2781
|
// to `return false` here, so `/plans-execute` answered "already executing"
|
|
@@ -2271,15 +2786,15 @@ export function resumeActiveExecution(ctx: ExtensionContext): boolean {
|
|
|
2271
2786
|
latchAuditThisSettle();
|
|
2272
2787
|
const chain = launchReviewRound(ctx);
|
|
2273
2788
|
if (chain) void chain.catch(() => { /* surfaced via the review messages */ });
|
|
2274
|
-
return true;
|
|
2789
|
+
return { resumed: true };
|
|
2275
2790
|
}
|
|
2276
|
-
if (!resumeGoalWaitIfPaused(ctx)) return false;
|
|
2791
|
+
if (!resumeGoalWaitIfPaused(ctx)) return { resumed: false };
|
|
2277
2792
|
const runtime = currentContinuationRuntime(ctx)!;
|
|
2278
2793
|
if (canWakeExecution(ctx, runtime)) {
|
|
2279
2794
|
runtime.handled = true;
|
|
2280
2795
|
sendContinuationWake(ctx, runtime);
|
|
2281
2796
|
}
|
|
2282
|
-
return true;
|
|
2797
|
+
return { resumed: true };
|
|
2283
2798
|
}
|
|
2284
2799
|
|
|
2285
2800
|
export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
@@ -2297,6 +2812,13 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
2297
2812
|
const residualNote = residualFindings.length > 0
|
|
2298
2813
|
? `\n\nRecorded findings that did not block completion: ${residualFindings.map((f) => `${f.id} (${f.severity})`).join(", ")} — see the execution-review round reports under the run directory.`
|
|
2299
2814
|
: "";
|
|
2815
|
+
// v0.9.3 (Q-2, round-1 F-004): an exhausted budget may complete WITH
|
|
2816
|
+
// unresolved high findings — the completion surface must say so instead of
|
|
2817
|
+
// claiming "execution review passed".
|
|
2818
|
+
const toleratedHighs = execution.audit.findings.filter((f) => f.severity === "high");
|
|
2819
|
+
const toleratedNote = toleratedHighs.length > 0
|
|
2820
|
+
? `\n\n⚠️ The review budget was exhausted (${budgetLabel(execution)}) before these high-severity finding(s) could be resolved: ${toleratedHighs.map((f) => f.id).join(", ")} — every verification check passed; the finding(s) remain in the round reports under the run directory.`
|
|
2821
|
+
: "";
|
|
2300
2822
|
const planPath = execution.planPath;
|
|
2301
2823
|
withExecutionCheckpoint(ctx, (cp) => applyExecutionCompleted(cp));
|
|
2302
2824
|
execution = null;
|
|
@@ -2306,7 +2828,9 @@ export async function completeExecution(ctx: ExtensionContext): Promise<void> {
|
|
|
2306
2828
|
messaging().sendMessage(
|
|
2307
2829
|
{
|
|
2308
2830
|
customType: "pi-plans-complete",
|
|
2309
|
-
content:
|
|
2831
|
+
content: toleratedHighs.length > 0
|
|
2832
|
+
? `**Plan complete (review budget exhausted).** ⚠️ \`${planPath}\` — all verification checks passed; ${toleratedHighs.length} high-severity finding(s) stayed unresolved.\n\n${summary}${residualNote}${toleratedNote}`
|
|
2833
|
+
: `**Plan complete!** ✅ \`${planPath}\` — execution review passed.\n\n${summary}${residualNote}`,
|
|
2310
2834
|
display: true,
|
|
2311
2835
|
},
|
|
2312
2836
|
{ triggerTurn: false },
|
|
@@ -2345,13 +2869,28 @@ export function executionContextMessage(ctx: ExtensionContext): string | null {
|
|
|
2345
2869
|
const highFindingsNote = unresolvedHighs.length > 0
|
|
2346
2870
|
? `\nExecution review round ${execution.audit.rounds} unresolved high-severity findings:\n${unresolvedHighs.map((f) => `- ${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("\n")}\nFix them, then re-close the affected tasks with evidence.`
|
|
2347
2871
|
: "";
|
|
2872
|
+
// v0.9.2: the wake must answer "is the review running?" explicitly. The
|
|
2873
|
+
// previous text left the executor waiting for a round that cannot start
|
|
2874
|
+
// while the tree is open (the exact stall this feature fixes).
|
|
2875
|
+
const blockedIds = blockedTaskIds(execution);
|
|
2876
|
+
const blockedRound = execution.blocked?.round ?? execution.audit.rounds;
|
|
2877
|
+
const outstanding = reviewOutstanding(execution);
|
|
2878
|
+
const reviewLine = execution.review.inFlight
|
|
2879
|
+
? `\nReview: round ${execution.audit.rounds + 1}/${budgetLabel(execution)} IS RUNNING (read-only reviewer verifying) — do not wait on it and do not re-close tasks for it.`
|
|
2880
|
+
: blockedIds.length > 0 && outstanding
|
|
2881
|
+
? `\nReview: NO round is running — the task tree is not terminal, so the review cannot start. It starts by itself the moment every task is terminal (round ${execution.audit.rounds + 1}/${budgetLabel(execution)}).`
|
|
2882
|
+
: "";
|
|
2883
|
+
const escalated = (execution.blocked?.escalatedRounds ?? 0) > 0;
|
|
2884
|
+
const blockedLine = blockedIds.length > 0 && outstanding
|
|
2885
|
+
? `\nBLOCKED — ${blockedIds.length} task(s) reopened by round ${blockedRound} are still open:\n${blockedIds.map((id) => `- ${id} (reopened by round ${blockedRound}) — close with \`plans_update_task\`: status "complete" with evidence, or "skipped" with a skipReason. Closing a child does NOT close its parent; a parent with an open child is not terminal.`).join("\n")}${escalated ? `\nThis is blocked wake ${execution.blocked?.escalatedRounds}/${STALL_MAX_ROUNDS}: if these tasks stay open, the watchdog pauses the run and a human has to resume it.` : ""}`
|
|
2886
|
+
: "";
|
|
2348
2887
|
return `[PI-PLANS EXECUTION — write access enabled]
|
|
2349
2888
|
Implement the accepted plan at ${execution.planPath} (tasks ${progress.done}/${progress.total}${execution.legacyPlan ? " · legacy I-### mapping" : ""} · VC ${vcDone}/${execution.items.length}).
|
|
2350
2889
|
|
|
2351
2890
|
Current wave ${currentWave} open tasks:
|
|
2352
2891
|
${waveList}
|
|
2353
2892
|
|
|
2354
|
-
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}
|
|
2893
|
+
Remaining open tasks (all waves): ${open.map((task) => task.id).join(", ") || "(none)"}.${rollbackNote}${highFindingsNote}${reviewLine}${blockedLine}
|
|
2355
2894
|
|
|
2356
2895
|
${graphLine}
|
|
2357
2896
|
|
|
@@ -2359,6 +2898,8 @@ Execution rules:
|
|
|
2359
2898
|
- Work through tasks in wave order (earlier waves first); within a wave, follow the listed dependency order. Wave grouping encodes which tasks could run in parallel — keep their file sets disjoint.
|
|
2360
2899
|
- Report progress ONLY through the \`plans_update_task\` tool: status "complete" with evidence (test command output / file paths), or "skipped" with a skipReason. One call per task; statuses are immutable once set.
|
|
2361
2900
|
- Close subtasks before their parent; a parent is auditable only when every child is terminal.
|
|
2901
|
+
- After a FAILED review round, every reopened task must be re-closed with fresh evidence — PARENTS INCLUDED. Closing a task's children does NOT close the task: a parent that still has an open child is not terminal, and the review cannot start until the whole tree is terminal.
|
|
2902
|
+
- NEVER wait for the review. While any task is open, the review is not running and nothing will re-close tasks for you: an open task is your work queue — close it with evidence or skip it with a skipReason.
|
|
2362
2903
|
- When every task is terminal, the independent execution reviewer verifies the plan's verification checks (${execution.items.map((item) => item.id).join(", ")}); failed checks roll their covered tasks back automatically.
|
|
2363
2904
|
- Simplest implementation that fully meets the task: no speculative abstractions, configuration, or indirection; keep components modular with clearly separated concerns.
|
|
2364
2905
|
- Architectural decisions are for the long term: no stopgaps. Remove the obsolete paths this change obsoletes.
|
|
@@ -2438,6 +2979,10 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
2438
2979
|
usage: snapshot.usage ?? { inToks: 0, outToks: 0 },
|
|
2439
2980
|
uiLanguage: resolveUiLanguage(ctx.cwd),
|
|
2440
2981
|
stall: { ...snapshot.stall, lastSnapshot: null },
|
|
2982
|
+
// v0.9.2: a snapshot taken by a pre-feature build carries the live failed
|
|
2983
|
+
// set but no blocker record; reconstruct it so the very first wake after
|
|
2984
|
+
// a /reload (the recovery path for an already-stuck run) names the tasks.
|
|
2985
|
+
blocked: snapshot.blocked ?? blockedFromFailedIds(snapshot.audit?.failed ?? [], snapshot.audit?.rounds ?? 0, tasks, items),
|
|
2441
2986
|
audit: {
|
|
2442
2987
|
rounds: snapshot.audit?.rounds ?? 0,
|
|
2443
2988
|
failed: snapshot.audit?.failed ?? [],
|
|
@@ -2445,8 +2990,16 @@ export async function restoreFromSession(ctx: ExtensionContext, entries: Session
|
|
|
2445
2990
|
findings: toReviewFindings(snapshot.audit?.findings),
|
|
2446
2991
|
running: false,
|
|
2447
2992
|
},
|
|
2448
|
-
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null },
|
|
2993
|
+
review: { attempts: 0, consecutiveDiscards: 0, inFlight: null, budgetAsking: false },
|
|
2449
2994
|
auditLatch: { auditedThisSettle: false, activity: 0 },
|
|
2995
|
+
// v0.9.3 (round-1 F-005): the snapshot's budget/counters survive a
|
|
2996
|
+
// /reload — an unset budget stays unset (the picker runs before round 1),
|
|
2997
|
+
// a legacy snapshot resolves to the 5-round bound via the audit counter.
|
|
2998
|
+
reviewBudget: resolveStoredBudget(snapshot.reviewBudget, snapshot.audit?.rounds ?? 0),
|
|
2999
|
+
reviewBudgetDefaulted: snapshot.reviewBudgetDefaulted === true,
|
|
3000
|
+
reviewRoundsTotal: snapshot.reviewRoundsTotal ?? 0,
|
|
3001
|
+
reviewCapExtension: snapshot.reviewCapExtension ?? 0,
|
|
3002
|
+
reviewNoProgress: snapshot.reviewNoProgress ?? undefined,
|
|
2450
3003
|
};
|
|
2451
3004
|
execution.stall.lastSnapshot = stallSnapshot();
|
|
2452
3005
|
resetContinuationRuntime(ctx);
|