pi-plans 0.8.1 → 0.8.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +13 -6
- package/references/state-and-config.md +1 -1
- package/src/auditor.ts +12 -5
- package/src/dashboard.ts +60 -9
- package/src/exec.ts +596 -43
- package/src/resume-command.ts +20 -6
- package/src/review-budget.ts +290 -0
- package/src/tasks.ts +53 -0
- package/src/ui-language.ts +38 -0
- package/src/workflow-state.ts +138 -3
- package/tests/dashboard.test.ts +98 -0
- package/tests/exec-review-loop.test.ts +632 -4
- package/tests/exec.test.ts +32 -5
- package/tests/fixtures/lattice-code-blocked/plan-v2-trimmed.md +57 -0
- package/tests/fixtures/lattice-code-blocked/state.json +158 -0
- package/tests/resume.test.ts +45 -1
- package/tests/review-budget.test.ts +201 -0
- package/tests/tasks.test.ts +45 -0
- package/tests/workflow-state.test.ts +187 -0
- package/tools/execute-plan.ts +12 -5
package/README.md
CHANGED
|
@@ -137,7 +137,7 @@ Planning artifacts live under `./.git/pi-plans/plans/YYYY-MM-DD-<topic>/` by def
|
|
|
137
137
|
| Visible Refiner overlay | Delegated reviewer subagents surface as a named public overlay in the TUI — one `Reviewer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
|
|
138
138
|
| Tracked execution | The current wave and remaining tasks are injected each turn; task progress is reported exclusively through the `plans_update_task` tool (status + evidence / skipReason, audit-only rollback); the task dashboard shows the tree live (compact aboveEditor widget, Ctrl+Shift+T expanded view with ✓/▸/~/· markers, width-adaptive); a stall watchdog pauses after three settled rounds without task-state change; the status bar shows lifecycle, `x/y` task progress, elapsed time, and token usage in real time |
|
|
139
139
|
| Multi-run workdirs (0.6.0) | Several pi sessions can plan concurrently in one workdir: the run registry derives from `runs/` (no shared pointer to race), each session binds to its run, and same-topic runs get suffixed artifact dirs. `/plans-abandon`, `/plans-execute`, and `/resume-plans` are binding-first and open a descriptive run-picker form when more than one candidate exists; `/plans` lists all runs (newest first, bound run marked) |
|
|
140
|
-
| Execution reviewer | When every task reaches a terminal state, the run status moves to `verifying` and an independent read-only reviewer verifies each `VC-###` check AND reports severity-graded `F-###` findings over the whole implemented change in a detached, overlay-visible round (Esc closes; Ctrl+Shift+R reopens the in-flight round). Finding ids are stable across rounds (absence from the newest report = resolved). A failed check or a high-severity finding opens one union fix round: mapped tasks roll back to pending (children cascade, skipped reopen; an unmapped high gets a plan task appended mechanically from the reviewer's proposed title), and the executor is woken exactly once with a findings summary plus the round-report path. The run completes
|
|
140
|
+
| Execution reviewer | When every task reaches a terminal state, the run status moves to `verifying` and an independent read-only reviewer verifies each `VC-###` check AND reports severity-graded `F-###` findings over the whole implemented change in a detached, overlay-visible round (Esc closes; Ctrl+Shift+R reopens the in-flight round). Finding ids are stable across rounds (absence from the newest report = resolved). A failed check or a high-severity finding opens one union fix round: mapped tasks roll back to pending (children cascade, skipped reopen; an unmapped high gets a plan task appended mechanically from the reviewer's proposed title), and the executor is woken exactly once with a findings summary plus the round-report path. The run completes when every check is affirmatively `pass` and no high finding remains; residual medium/low findings are summarized at completion, and an exhausted budget completes with any unresolved high findings disclosed by id. The review budget is a per-run choice (1/2/3/5/unlimited, default 3) asked exactly once — when every task is terminal, right before round 1 — and stored in the checkpoint: a numeric budget bounds committed rounds, while `unlimited` runs until no high finding remains behind a no-progress valve (three consecutive identical outcomes) and a run-cumulative hard cap of 50 committed rounds. Exhaustion pauses in every mode, and only an explicit `/plans-execute` confirmation re-opens the budget picker and grants a fresh budget (unresolved findings survive the renewal; ordinary input and restores never refill). Checks with all-skipped coverage pass; checks covering no task never audit |
|
|
141
141
|
| Execution handoff | The accepted plan executes in the current session after explicit approval (never auto-completed); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; 0.6.0 in-flight runs resume compatibly (delegated-executor orphans re-approve, paused executions rebuild from the task tree) |
|
|
142
142
|
| Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current task, task ids, and remaining `VC-###` checks into the VCC sections. Proactive triggers and model-generated summary paths are removed. |
|
|
143
143
|
| Planning-phase compaction | During `run.status=planning` with no active execution, pi-plans maps active run, artifact directory, latest plan path from session entries, and observed current-I markers into the VCC sections. Without an active planning run, compaction returns to Pi core. Additionally, creating a new run (`plans start-run`) proactively requests one pre-plan VCC compaction and resumes planning with a hidden message (default on; `prePlanCompact:false` disables). |
|
|
@@ -310,7 +310,7 @@ The plan is the contract. Refinement converges on scope while nothing is writabl
|
|
|
310
310
|
|
|
311
311
|
**What can Auto-complete decide on my behalf?**
|
|
312
312
|
|
|
313
|
-
Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, the independent execution reviewer verifies every check; the
|
|
313
|
+
Planning and refinement choices only (the recommended option). Choosing Auto-complete enables the recommended answer for later eligible planning questions in the current run and the extension continues the planning turn when the model stops early. Use `/plans-autocomplete-stop` to take back control. It is never offered for execution approval, installs, publishing, deployment, merge, push, or credentials — those questions stop and wait for you. After execution completes, the independent execution reviewer verifies every check; the per-run review budget (1/2/3/5/unlimited, default 3, chosen when the task tree first goes terminal) pauses the run for review in every mode when exhausted, and only an explicit `/plans-execute` confirmation re-opens the budget picker and grants a fresh budget.
|
|
314
314
|
|
|
315
315
|
**Where does all the state live?**
|
|
316
316
|
|
package/package.json
CHANGED
|
@@ -131,16 +131,20 @@ When the user picks `✓ Accept PLAN_vN and execute it now` in the merged questi
|
|
|
131
131
|
|
|
132
132
|
- every agent turn is injected with the current wave's open tasks, the remaining task list, verification-check summary, and execution rules (wave order, `plans_update_task` reporting with status + evidence / skipReason, subprocess polling backoff 5s -> 10s -> 20s -> 40s -> 80s then keep polling at 80s, no stopgaps, dependency and library discipline, minimum tests);
|
|
133
133
|
- task progress flows exclusively through the `plans_update_task` tool: one call per task closing it as `complete` (with evidence) or `skipped` (with skipReason); closed statuses are immutable outside the audit-authorized rollback channel; subtasks close before their parent;
|
|
134
|
-
- the task dashboard tracks the whole tree live: a compact aboveEditor widget (current task ▸, progress bar, ✓/· counts, VC pass count, wave indicator, pause state, audit-outcome line, unresolved-findings line visible in both the repairing and verifying phases) and the Ctrl+Shift+T expanded tree view (✓/▸/~/·/↺ markers, current-task anchor, VC list with audit state, width-adaptive layout). `↺` marks a task that was completed and then rolled back by a failed audit: it is open again but still shows the evidence from its previous attempt;
|
|
135
|
-
- a
|
|
134
|
+
- the task dashboard tracks the whole tree live: a compact aboveEditor widget (current task ▸, progress bar, ✓/· counts, VC pass count, wave indicator, pause state, audit-outcome line, unresolved-findings line visible in both the repairing and verifying phases, and a `⊘ blocked` row naming the open rolled-back tasks while the review is blocked) and the Ctrl+Shift+T expanded tree view (✓/▸/~/·/↺ markers, current-task anchor, VC list with audit state, width-adaptive layout). `↺` marks a task that was completed and then rolled back by a failed audit: it is open again but still shows the evidence from its previous attempt;
|
|
135
|
+
- while a failed round's reopened tasks are still open, the wake message states explicitly that NO review round is running (the tree is not terminal, so the review cannot start) and carries a `BLOCKED` block naming every open rolled-back task, its provenance (`reopened by round N`), and the required `plans_update_task` action — closing a task's children does not close the task itself. The blocker is persisted in the checkpoint (`execution.blocked`), so the pause reason, the dashboard row, and the `/resume-plans` brief can name it after a restart;
|
|
136
|
+
- a stall watchdog pauses execution after three consecutive settled rounds without any task-status change (genuine user input or `/plans-execute` resumes without losing progress). A round in which the agent ran successful tool calls counts as progress even with no status change, so investigating the codebase is never mistaken for a dead agent. Blocked raises escalate on their OWN ladder (`execution.blocked.escalatedRounds`, which tool activity does not reset but which DOES reset to zero whenever the blocker set shrinks — closing one reopened task per settled round is progress, never a reason to harden or pause): the wake wording hardens on each consecutive no-progress blocked wake, ONE visible `pi-plans-exec-blocked` system line is emitted per escalation level, and the third consecutive no-progress blocked wake pauses the run with the blocker as the reason (`blocked: review round N cannot start — …`) instead of the watchdog's own metric. A resume (ordinary input or `/plans-execute`) grants a fresh blocked ladder without refilling the review budget;
|
|
136
137
|
- when every task reaches a terminal state, the run status moves to `verifying` and an independent read-only execution reviewer (`agents/execution-reviewer.md`) verifies each check against the worktree in a detached round whose progress renders in a dedicated overlay (Esc closes it; Ctrl+Shift+R reopens the in-flight round). Each check gets one of three verdicts:
|
|
137
138
|
- `pass` — the evidence establishes the check's condition;
|
|
138
139
|
- `fail` — the condition is demonstrably not met; the covered tasks roll back to pending (children cascade, skipped tasks reopen, the `evidence` of the previous attempt is retained while `skipReason` is cleared) and the audit report is injected;
|
|
139
|
-
- `undeterminable` — the reviewer could not reach a conclusion. This is never a failure and never a completion: nothing rolls back, the loop self-schedules the retry (no agent wake), and each retry counts
|
|
140
|
+
- `undeterminable` — the reviewer could not reach a conclusion. This is never a failure and never a completion: nothing rolls back, the loop self-schedules the retry (no agent wake), and each retry counts as a committed round against the run's review budget. An all-undeterminable round that also reports a high-severity finding still wakes the executor — a finding is actionable independent of verdict evidence;
|
|
140
141
|
- `F-###` findings — every round also reports severity-graded implementation findings (`high | medium | low`) over the whole implemented change, not just what the checks cover. Ids are stable: the brief lists the previous round's unresolved findings and the reviewer reuses their ids verbatim while the problem persists; absence from the newest round's report is the resolution signal. A `high` finding (or a `fail` verdict) opens ONE union fix round: mapped tasks roll back to pending exactly like a failed check's coverage (children cascade, evidence retained), and a `high` whose `tasks: none` names no owner gets a plan task appended mechanically from the reviewer's `proposed-task` title — the reviewer itself stays read-only. The executor is woken exactly once with a findings summary plus the round-report path. A pure finding-driven rollback deliberately keeps earlier VC passes (the findings channel re-examines the repaired work next round); only a `fail`-driven rollback invalidates the checks covering the reopened tasks. Unresolved findings persist across checkpoint restores, session restores, and fresh budget grants.
|
|
141
142
|
|
|
142
143
|
The run completes only when every pending check is affirmatively `pass` AND the newest round reports no high-severity finding; residual `medium`/`low` findings are summarized in the completion message and stay recorded in the round reports. A partial round credits the checks that passed and leaves the rest for the next round. Checks whose covered tasks are all skipped pass as skipped-pass; checks covering no task never enter the audit; checks already satisfied in an earlier round are neither re-briefed nor re-judged;
|
|
143
|
-
- the round budget is
|
|
144
|
+
- the round budget is a PER-RUN choice, resolved exactly once — right before round 1, when every task is terminal and the review is genuinely owed. An interactive session shows the budget panel (1 / 2 / 3 / 5 / unlimited, preselected on a re-pick); every other case (no UI, headless/print/json, auto-approve, RPC without menus) applies the default 3 with one visible note, and the checkpoint records `reviewBudgetDefaulted` so the dashboard and the resume brief mark it `(default)`. The value lives in the run checkpoint (`execution.reviewBudget`), never in the plan file, and picking is never a per-round question;
|
|
145
|
+
- a numeric budget counts COMMITTED rounds (pass, fail, or undeterminable; discarded fingerprint-mismatch attempts and cancellations burn nothing; two consecutive discards commit as one undeterminable round) and resets per grant;
|
|
146
|
+
- `unlimited` runs until no high finding remains, guarded by two valves: the no-progress valve pauses after three consecutive committed rounds with an identical outcome signature (sorted failed checks + undeterminable checks + high-finding ids, so a round that never produces a report — spawn failure or discard synthesis — still counts), and a run-cumulative hard cap of 50 committed rounds (`execution.reviewRoundsTotal`, never reset) pauses the run. An explicit `/plans-execute` grant that lands on `unlimited` lifts the hard cap by another 50 rounds (`execution.reviewCapExtension`);
|
|
147
|
+
- exhaustion pauses the run in EVERY mode — interactive and auto-approve/headless alike — with an in-band `pi-plans-review-paused` message naming the real budget, EXCEPT when every verification check is already satisfied: then the run completes even with unresolved high findings, which the completion message discloses by id (the findings stay in the round reports and the checkpoint). Ordinary user input and session restores never lift a pause or refill the budget. The only fresh-budget surface is `/plans-execute`: its confirmation re-opens the budget panel with the current value preselected and resets the per-grant round counter, the no-progress valve, and (only for an `unlimited` pick) extends the hard cap; Esc keeps the run paused. A pre-feature checkpoint that already spent rounds keeps its legacy bound of five committed rounds instead of being cut to the new default mid-flight. The watchdog counter is rebased by real tool activity;
|
|
144
148
|
- execution-phase compaction is handled only when Pi core emits manual `/compact`, threshold, or overflow events; summaries are deterministic VCC-style summaries, include session-derived plan/current-task/checklist context, use smart tail keep and `keep:N`, and never call a model;
|
|
145
149
|
- the read-only guard lifts: full write access returns;
|
|
146
150
|
- the run status moves to `executing`, then `verifying` while the review loop owns the run, then `done` when every check passes (a failed round rolls its tasks back and returns the run to `executing` for repair);
|
|
@@ -157,8 +161,11 @@ session is idle, and neither pending input nor compaction owns continuation.
|
|
|
157
161
|
Each eligible settled cycle can send at most one hidden custom message with
|
|
158
162
|
the current execution rules. Tool `turn_end` events only update usage; they
|
|
159
163
|
never prequeue continuation reminders. The stall watchdog counts settled
|
|
160
|
-
cycles without task-state change (threshold 3)
|
|
161
|
-
|
|
164
|
+
cycles without task-state change (threshold 3); while a rolled-back blocker
|
|
165
|
+
keeps the review from starting, the blocked ladder escalates the wake first
|
|
166
|
+
(naming the open tasks and stating that no review round is running) and only
|
|
167
|
+
its third consecutive raise pauses — a blocked pause names the blocker rather
|
|
168
|
+
than the watchdog metric. User interruption and final model errors pause too. Genuine interactive/RPC
|
|
162
169
|
user input or `/plans-execute` resumes a paused active execution without
|
|
163
170
|
losing task progress; extension input cannot unpause it. New-plan handoffs
|
|
164
171
|
and the `execute_plan` tool still require explicit approval. Print/JSON
|
|
@@ -207,7 +207,7 @@ Each run may carry a `checkpoint.json` — the durable, cross-session workflow s
|
|
|
207
207
|
|
|
208
208
|
Rules:
|
|
209
209
|
|
|
210
|
-
- Validation is explicit: unknown schema versions, malformed shapes, and unexpected keys are rejected; missing and corrupt checkpoints are distinct, and corrupt files are never silently overwritten. Keys added by a later version (`execution.stallRounds`, `execution.audit.undeterminable`, `execution.audit.findings`, `execution.planAmended`) are optional on read, so checkpoints written before them keep loading. `execution.planAmended` is set when the review loop mechanically appended finding tasks to the approved plan; the checkpoint's plan identity was re-stamped to the amended digest at that moment while the approval record keeps the original.
|
|
210
|
+
- Validation is explicit: unknown schema versions, malformed shapes, and unexpected keys are rejected; missing and corrupt checkpoints are distinct, and corrupt files are never silently overwritten. Keys added by a later version (`execution.stallRounds`, `execution.blocked`, `execution.reviewBudget`, `execution.reviewBudgetDefaulted`, `execution.reviewRoundsTotal`, `execution.reviewCapExtension`, `execution.reviewNoProgress`, `execution.audit.undeterminable`, `execution.audit.findings`, `execution.planAmended`) are optional on read, so checkpoints written before them keep loading. `execution.planAmended` is set when the review loop mechanically appended finding tasks to the approved plan; the checkpoint's plan identity was re-stamped to the amended digest at that moment while the approval record keeps the original. `execution.blocked` (`{ rolledBack, tasks, round, escalatedRounds, since }`) records the newest failed round's rollback set — captured when the round commits, because the reopen helpers mutate the task tree and report only flipped nodes — plus the still-open tasks among it; it drives the `BLOCKED` wake block, the blocked escalation ladder, the pause reason, the dashboard row, and the resume brief. It is deleted (not left stale) when the last blocker closes, when the run stops, and when the audit passes. `execution.reviewBudget` is the per-run execution-review budget: a positive integer round count or the literal `"unlimited"` (absent = not chosen yet, or written by a pre-0.9.3 build — a checkpoint that already spent rounds resolves to the legacy bound of five). `execution.reviewBudgetDefaulted` marks a budget that came from the no-UI fallback rather than a user pick (the dashboard and resume brief annotate it `(default)`). `execution.reviewRoundsTotal` counts committed review rounds across the whole run and never resets; together with `execution.reviewCapExtension` (lifted by 50 rounds on every explicit `/plans-execute` grant that lands on `unlimited`) it bounds the `unlimited` budget with a run-cumulative hard cap. `execution.reviewNoProgress` (`{ key, streak }`) is the unlimited budget's no-progress valve state: the signature of the newest committed round's outcome (sorted failed checks + undeterminable checks + high-finding ids) plus its consecutive-run count. All five keys are dropped by a stop and by completion (a migration carries them, since they describe the user's choice and the run's spend, not the invalidated code state).
|
|
211
211
|
- Writes are atomic with monotonic revisions; writers may require ownership (token + generation) or an expected revision.
|
|
212
212
|
- Model-driven boundaries (plan written, review consolidated, termination condition recorded, implementation round finished, completed) go through the whitelisted `plans record-checkpoint` action, which enforces state-machine preconditions — it cannot set execution approval, mark VCs passed, or forge terminal states.
|
|
213
213
|
- `ask_choice` accepts `questionId`/`purpose`; a pending question is durable before the panel opens and the answer before it returns. When a crash leaves a question both answered (ledger) and pending (checkpoint), the answered entry wins.
|
package/src/auditor.ts
CHANGED
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
* reaches a terminal state, an independent read-only subagent verifies the
|
|
4
4
|
* plan's verification checks against the worktree. Failed checks roll their
|
|
5
5
|
* covered tasks back to pending (exclusively inside the review flow). The
|
|
6
|
-
* loop is bounded
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* loop is bounded by the per-run review budget (v0.9.3: `ReviewBudget` in
|
|
7
|
+
* ./review-budget.ts, chosen right before round 1 and stored in the run
|
|
8
|
+
* checkpoint); exhaustion pauses the run for the user in every mode
|
|
9
|
+
* (fail-closed, never a hang and never a silent stop) — only an explicit
|
|
10
|
+
* /plans-execute confirmation grants a fresh budget.
|
|
10
11
|
*/
|
|
11
12
|
|
|
12
13
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
@@ -16,7 +17,13 @@ import { auditableChecks, flattenTaskViews, skippedPassCheckIds, type TaskView }
|
|
|
16
17
|
import { normalizeTaskId, type CheckItem } from "./plan.ts";
|
|
17
18
|
import { messaging } from "./messaging.ts";
|
|
18
19
|
|
|
19
|
-
/**
|
|
20
|
+
/** Legacy fixed cap (v0.8–v0.9.2): the review-round budget is now a per-run
|
|
21
|
+
* choice (`ReviewBudget` in ./review-budget.ts, default 3, pickable 1/2/3/5/
|
|
22
|
+
* unlimited). This constant survives ONLY for checkpoints and tests written
|
|
23
|
+
* against the old fixed budget — a pre-feature checkpoint that already spent
|
|
24
|
+
* rounds keeps the 5-round bound (`LEGACY_REVIEW_MAX_ROUNDS`). It no longer
|
|
25
|
+
* gates any round.
|
|
26
|
+
* @deprecated use `resolveStoredBudget` / `budgetExhausted` from ./review-budget.ts */
|
|
20
27
|
export const REVIEW_MAX_ROUNDS = 5;
|
|
21
28
|
|
|
22
29
|
/** Legacy alias for one release: checkpoints and old builds still know this name. */
|
package/src/dashboard.ts
CHANGED
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
*/
|
|
17
17
|
|
|
18
18
|
import type { CheckItem } from "./plan.ts";
|
|
19
|
-
import {
|
|
19
|
+
import { LEGACY_REVIEW_MAX_ROUNDS, formatReviewBudget, unlimitedHardCapCeiling, type ReviewBudget } from "./review-budget.ts";
|
|
20
20
|
import { truncateToWidth, visibleWidth } from "./refine-ui-helpers.ts";
|
|
21
21
|
import {
|
|
22
22
|
allTasksTerminal,
|
|
@@ -49,6 +49,21 @@ export interface DashboardModel {
|
|
|
49
49
|
/** v0.9: unresolved findings from the newest committed review round
|
|
50
50
|
* (stable ids). High entries block completion; the rest are recorded. */
|
|
51
51
|
findings: Array<{ id: string; severity: string; note: string; taskIds: string[] }>;
|
|
52
|
+
/** v0.9.2: tasks that keep the review from starting — the newest failed
|
|
53
|
+
* round's rollback set intersected with the still-open tasks. Empty = the
|
|
54
|
+
* review is not blocked by open work. */
|
|
55
|
+
blockedTasks: string[];
|
|
56
|
+
/** Round that reopened those tasks (shown as `reopened by round N`). */
|
|
57
|
+
blockedRound: number | null;
|
|
58
|
+
/** v0.9.3: the per-run execution-review budget (rounds or `unlimited`).
|
|
59
|
+
* Null/absent = not chosen yet; `reviewBudgetDefaulted` marks the no-UI
|
|
60
|
+
* fallback so the expanded view can annotate it `(default)`. */
|
|
61
|
+
reviewBudget?: ReviewBudget | null;
|
|
62
|
+
/** v0.9.3: committed rounds across the whole run + granted extension —
|
|
63
|
+
* the unlimited budget's `n/50` progress. */
|
|
64
|
+
reviewRoundsTotal?: number;
|
|
65
|
+
reviewCapExtension?: number;
|
|
66
|
+
reviewBudgetDefaulted?: boolean;
|
|
52
67
|
startedAt: string;
|
|
53
68
|
usage: { inToks: number; outToks: number };
|
|
54
69
|
}
|
|
@@ -177,20 +192,32 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
|
|
|
177
192
|
const nextRound = (model.auditRounds ?? 0) + 1;
|
|
178
193
|
const highCount = model.findings.filter((f) => f.severity === "high").length;
|
|
179
194
|
const auditLine = model.reviewRunning
|
|
180
|
-
? `review: round ${nextRound}/${
|
|
195
|
+
? `review: round ${nextRound}/${budgetLabelOf(model)} running — read-only reviewer verifying`
|
|
181
196
|
: model.auditFailed.length > 0
|
|
182
197
|
? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
|
|
183
198
|
: highCount > 0
|
|
184
|
-
? `review: round ${nextRound}/${
|
|
199
|
+
? `review: round ${nextRound}/${budgetLabelOf(model)} — ${highCount} high finding(s) unresolved`
|
|
185
200
|
: model.auditUndeterminable.length > 0
|
|
186
201
|
? `audit: ${model.auditUndeterminable.length} check(s) undeterminable — verdict unreadable`
|
|
187
202
|
: owed
|
|
188
|
-
? `review: round ${nextRound}/${
|
|
203
|
+
? `review: round ${nextRound}/${budgetLabelOf(model)} — verdict pending`
|
|
189
204
|
: model.auditRounds !== null
|
|
190
205
|
? "audit complete ✓"
|
|
191
206
|
: "all tasks terminal — audit pending";
|
|
192
207
|
lines.push(boxRow("│", ` ${clip(auditLine, inner - 2)}`, " ", width));
|
|
193
208
|
}
|
|
209
|
+
// v0.9.2: the blocker row renders in BOTH phases — a paused run sees it
|
|
210
|
+
// below the pause summary (which may be clipped at narrow widths), and a
|
|
211
|
+
// live run sees exactly what keeps the review from starting.
|
|
212
|
+
if (model.blockedTasks.length > 0) {
|
|
213
|
+
const provenance = model.blockedRound !== null ? ` (reopened by round ${model.blockedRound})` : "";
|
|
214
|
+
// Provenance once per row keeps the id list readable and short enough to
|
|
215
|
+
// survive a 100-column panel without dropping the instruction.
|
|
216
|
+
const rowText = narrow
|
|
217
|
+
? `⊘ blocked: ${model.blockedTasks.join(", ")}${provenance}`
|
|
218
|
+
: `⊘ blocked: ${model.blockedTasks.join(", ")}${provenance} — close with plans_update_task`;
|
|
219
|
+
lines.push(boxRow("│", ` ${clip(rowText, inner - 2)}`, " ", width));
|
|
220
|
+
}
|
|
194
221
|
if (!narrow && model.auditFailed.length > 0) {
|
|
195
222
|
lines.push(boxRow("│", ` ✗ ${clip(model.auditFailed.join(", "), inner - 3)}`, " ", width));
|
|
196
223
|
}
|
|
@@ -220,6 +247,22 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
|
|
|
220
247
|
return clampLines(painted, width);
|
|
221
248
|
}
|
|
222
249
|
|
|
250
|
+
/** v0.9.3: the budget denominator as shown in every review line (`3`, `∞`).
|
|
251
|
+
* A model without the field (legacy fixtures) reads as the legacy 5. */
|
|
252
|
+
function budgetLabelOf(model: DashboardModel): string {
|
|
253
|
+
return formatReviewBudget(model.reviewBudget ?? LEGACY_REVIEW_MAX_ROUNDS);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/** v0.9.3: unlimited budgets render their run-cumulative hard-cap progress. */
|
|
257
|
+
function hardCapOf(model: DashboardModel): string | null {
|
|
258
|
+
if ((model.reviewBudget ?? null) !== "unlimited") return null;
|
|
259
|
+
const ceiling = unlimitedHardCapCeiling({
|
|
260
|
+
reviewRoundsTotal: model.reviewRoundsTotal ?? 0,
|
|
261
|
+
reviewCapExtension: model.reviewCapExtension ?? 0,
|
|
262
|
+
});
|
|
263
|
+
return `${model.reviewRoundsTotal ?? 0}/${ceiling}`;
|
|
264
|
+
}
|
|
265
|
+
|
|
223
266
|
function progressBar(done: number, total: number, width = 12): string {
|
|
224
267
|
if (total <= 0) return "";
|
|
225
268
|
const filled = Math.round((done / total) * width);
|
|
@@ -230,7 +273,7 @@ export function deriveDashboardModel(
|
|
|
230
273
|
topic: string,
|
|
231
274
|
tasks: TaskView[],
|
|
232
275
|
checklist: CheckItem[],
|
|
233
|
-
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; findings?: Array<{ id: string; severity: string; note: string; taskIds: string[] }>; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
276
|
+
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; findings?: Array<{ id: string; severity: string; note: string; taskIds: string[] }>; blockedTasks?: string[]; blockedRound?: number | null; reviewBudget?: ReviewBudget | null; reviewRoundsTotal?: number; reviewCapExtension?: number; reviewBudgetDefaulted?: boolean; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
234
277
|
): DashboardModel {
|
|
235
278
|
return {
|
|
236
279
|
topic,
|
|
@@ -243,6 +286,12 @@ export function deriveDashboardModel(
|
|
|
243
286
|
auditUndeterminable: extra?.auditUndeterminable ?? [],
|
|
244
287
|
reviewRunning: extra?.reviewRunning ?? false,
|
|
245
288
|
findings: extra?.findings ?? [],
|
|
289
|
+
blockedTasks: extra?.blockedTasks ?? [],
|
|
290
|
+
blockedRound: extra?.blockedRound ?? null,
|
|
291
|
+
reviewBudget: extra?.reviewBudget ?? null,
|
|
292
|
+
reviewRoundsTotal: extra?.reviewRoundsTotal ?? 0,
|
|
293
|
+
reviewCapExtension: extra?.reviewCapExtension ?? 0,
|
|
294
|
+
reviewBudgetDefaulted: extra?.reviewBudgetDefaulted ?? false,
|
|
246
295
|
startedAt: extra?.startedAt ?? new Date().toISOString(),
|
|
247
296
|
usage: extra?.usage ?? { inToks: 0, outToks: 0 },
|
|
248
297
|
};
|
|
@@ -258,10 +307,11 @@ export function formatDashboardSummaryLine(model: DashboardModel): string {
|
|
|
258
307
|
// high count — visible in BOTH phases (executing repair and verifying), so
|
|
259
308
|
// convergence is legible exactly while the executor is fixing.
|
|
260
309
|
const highs = model.findings.filter((f) => f.severity === "high").length;
|
|
261
|
-
const audit = model.auditRounds !== null ? ` · review r${model.auditRounds}/${
|
|
310
|
+
const audit = model.auditRounds !== null ? ` · review r${model.auditRounds}/${budgetLabelOf(model)}` : "";
|
|
262
311
|
const highToken = highs > 0 ? ` · ${highs} high` : "";
|
|
263
312
|
const pause = model.paused ? " · ⏸ paused" : "";
|
|
264
|
-
|
|
313
|
+
const blocked = model.blockedTasks.length > 0 ? " · ⊘ blocked" : "";
|
|
314
|
+
return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${highToken}${pause}${blocked}`;
|
|
265
315
|
}
|
|
266
316
|
|
|
267
317
|
/** Expanded tree view lines (Ctrl+Shift+T overlay). Wide layout from 96 cols. */
|
|
@@ -302,9 +352,10 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
|
|
|
302
352
|
lines.push(` ${mark} ${f.id} (${f.severity}${f.taskIds.length ? `, ${f.taskIds.join(", ")}` : ""}): ${wide ? clip(f.note, 72) : clip(f.note, 40)}`);
|
|
303
353
|
}
|
|
304
354
|
}
|
|
355
|
+
const budgetNote = `${budgetLabelOf(model)}${model.reviewBudgetDefaulted ? " (default)" : ""}${hardCapOf(model) ? ` · cap ${hardCapOf(model)}` : ""}`;
|
|
305
356
|
if (model.reviewRunning) {
|
|
306
357
|
lines.push("");
|
|
307
|
-
lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${
|
|
358
|
+
lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${budgetNote} running`);
|
|
308
359
|
} else if (model.auditRounds !== null) {
|
|
309
360
|
lines.push("");
|
|
310
361
|
const highIds = model.findings.filter((f) => f.severity === "high").map((f) => f.id);
|
|
@@ -315,7 +366,7 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
|
|
|
315
366
|
: model.auditUndeterminable.length > 0
|
|
316
367
|
? ` — undeterminable: ${model.auditUndeterminable.join(", ")}`
|
|
317
368
|
: " — passed ✓";
|
|
318
|
-
lines.push(`Execution review: round ${model.auditRounds}${verdict}`);
|
|
369
|
+
lines.push(`Execution review: round ${model.auditRounds}/${budgetNote}${verdict}`);
|
|
319
370
|
}
|
|
320
371
|
const painted = theme ? lines.map((line) => theme.fg("muted", line)) : lines;
|
|
321
372
|
return clampLines(painted, width);
|