pi-plans 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/agents/execution-reviewer.md +62 -10
- package/package.json +2 -2
- package/references/pi-planning-workflow.md +4 -3
- package/references/state-and-config.md +2 -2
- package/src/auditor.ts +158 -16
- package/src/dashboard.ts +47 -15
- package/src/exec.ts +282 -77
- package/src/refine-ui.ts +21 -5
- package/src/resume-command.ts +7 -1
- package/src/tasks.ts +23 -0
- package/src/workflow-state.ts +80 -6
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +185 -1
- package/tests/dashboard.test.ts +67 -1
- package/tests/exec-review-loop.test.ts +400 -7
- package/tests/refine-ui.test.ts +25 -2
- package/tests/workflow-state.test.ts +40 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/refine.ts +22 -3
package/README.md
CHANGED
|
@@ -137,7 +137,7 @@ Planning artifacts live under `./.git/pi-plans/plans/YYYY-MM-DD-<topic>/` by def
|
|
|
137
137
|
| Visible Refiner overlay | Delegated reviewer subagents surface as a named public overlay in the TUI — one `Reviewer` panel with per-lane tool progress, full streaming transcript with follow-bottom scroll, Tab-pane focus, retention until the user presses `Esc` after completion, and clean cancelled/timed-out vs completed states. `reviewers: 3` renders three equal-height panes inside the same overlay |
|
|
138
138
|
| Tracked execution | The current wave and remaining tasks are injected each turn; task progress is reported exclusively through the `plans_update_task` tool (status + evidence / skipReason, audit-only rollback); the task dashboard shows the tree live (compact aboveEditor widget, Ctrl+Shift+T expanded view with ✓/▸/~/· markers, width-adaptive); a stall watchdog pauses after three settled rounds without task-state change; the status bar shows lifecycle, `x/y` task progress, elapsed time, and token usage in real time |
|
|
139
139
|
| Multi-run workdirs (0.6.0) | Several pi sessions can plan concurrently in one workdir: the run registry derives from `runs/` (no shared pointer to race), each session binds to its run, and same-topic runs get suffixed artifact dirs. `/plans-abandon`, `/plans-execute`, and `/resume-plans` are binding-first and open a descriptive run-picker form when more than one candidate exists; `/plans` lists all runs (newest first, bound run marked) |
|
|
140
|
-
| Execution reviewer | When every task reaches a terminal state, the run status moves to `verifying` and an independent read-only reviewer verifies each `VC-###` check in a detached, overlay-visible round (Esc closes; Ctrl+Shift+R reopens the in-flight round)
|
|
140
|
+
| Execution reviewer | When every task reaches a terminal state, the run status moves to `verifying` and an independent read-only reviewer verifies each `VC-###` check AND reports severity-graded `F-###` findings over the whole implemented change in a detached, overlay-visible round (Esc closes; Ctrl+Shift+R reopens the in-flight round). Finding ids are stable across rounds (absence from the newest report = resolved). A failed check or a high-severity finding opens one union fix round: mapped tasks roll back to pending (children cascade, skipped reopen; an unmapped high gets a plan task appended mechanically from the reviewer's proposed title), and the executor is woken exactly once with a findings summary plus the round-report path. The run completes only when every check is affirmatively `pass` and no high finding remains; residual medium/low findings are summarized at completion. Five committed rounds bound the loop — exhaustion pauses in every mode, and only an explicit `/plans-execute` confirmation grants a fresh budget (unresolved findings survive the renewal; ordinary input and restores never refill). Checks with all-skipped coverage pass; checks covering no task never audit |
|
|
141
141
|
| Execution handoff | The accepted plan executes in the current session after explicit approval (never auto-completed); legacy `I-###` plans parse through the compatibility mapping with an upgrade notice; 0.6.0 in-flight runs resume compatibly (delegated-executor orphans re-approve, paused executions rebuild from the task tree) |
|
|
142
142
|
| Execution-phase compaction | Pi core owns scheduling; pi-plans maps the active plan path, current task, task ids, and remaining `VC-###` checks into the VCC sections. Proactive triggers and model-generated summary paths are removed. |
|
|
143
143
|
| Planning-phase compaction | During `run.status=planning` with no active execution, pi-plans maps active run, artifact directory, latest plan path from session entries, and observed current-I markers into the VCC sections. Without an active planning run, compaction returns to Pi core. Additionally, creating a new run (`plans start-run`) proactively requests one pre-plan VCC compaction and resumes planning with a hidden message (default on; `prePlanCompact:false` disables). |
|
|
@@ -1,32 +1,41 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: pi-plans-execution-reviewer
|
|
3
|
-
description: Read-only execution reviewer for pi-plans; verifies an implemented worktree against the plan's verification checks and
|
|
3
|
+
description: Read-only execution reviewer for pi-plans; verifies an implemented worktree against the plan's verification checks and reports severity-graded implementation findings that drive the fix loop.
|
|
4
4
|
tools: read, grep, find, ls
|
|
5
5
|
---
|
|
6
6
|
|
|
7
|
-
You are the execution reviewer in the pi-plans workflow.
|
|
8
|
-
whether an already-implemented worktree satisfies the
|
|
9
|
-
accepted plan
|
|
7
|
+
You are the execution reviewer in the pi-plans workflow. Each review round has
|
|
8
|
+
two jobs: decide whether an already-implemented worktree satisfies the
|
|
9
|
+
verification checks of an accepted plan, and report implementation findings —
|
|
10
|
+
defects you can point at in the repository — that the executor will fix before
|
|
11
|
+
the next round. A round is useful only when both outputs are present.
|
|
10
12
|
|
|
11
13
|
Rules:
|
|
12
14
|
|
|
13
15
|
- Perform read-only analysis. Never edit, write, or delete any file, never commit, never push, never spawn subagents.
|
|
14
16
|
- Verify against the actual repository using your read tools before judging. A check passes only on evidence you actually inspected.
|
|
15
17
|
- You are auditing a worktree that is already written. "The file is missing" is a finding to report, not a reason to stay silent.
|
|
16
|
-
- You do not fix anything
|
|
18
|
+
- You do not fix anything. You report verdicts and findings; the fix loop acts on them.
|
|
17
19
|
|
|
18
20
|
## Output contract
|
|
19
21
|
|
|
20
|
-
Output
|
|
21
|
-
|
|
22
|
+
Output exactly two sections, in this order.
|
|
23
|
+
|
|
24
|
+
### 1. Verification verdicts
|
|
25
|
+
|
|
26
|
+
One section per check, in the order given by the brief. The preferred shape
|
|
27
|
+
puts the id and the verdict on one line (the runner reads this form first):
|
|
22
28
|
|
|
23
29
|
- `VC-###` — verdict: pass | fail | undeterminable; evidence: <repo path/command or recorded output proving it>; note: <one line>.
|
|
24
30
|
|
|
31
|
+
A heading-style section is also read: start it with the id (`### VC-###` or a
|
|
32
|
+
bullet that names the check) and keep that check's `verdict:` line inside the
|
|
33
|
+
section. Whichever form you choose, use ONE form for the whole report and
|
|
34
|
+
never let one check's verdict drift into another's section.
|
|
35
|
+
|
|
25
36
|
Emit **every** check the brief lists, in the brief's order. Never omit a check,
|
|
26
37
|
never merge two checks into one section, never invent a check that is not listed.
|
|
27
38
|
|
|
28
|
-
## Choosing the verdict
|
|
29
|
-
|
|
30
39
|
- `pass` — you inspected the evidence and it establishes the check's condition.
|
|
31
40
|
- `fail` — you inspected the evidence and the check's condition is **demonstrably** not met. Use this only when you can point at the specific thing that breaks the condition.
|
|
32
41
|
- `undeterminable` — you could **not** reach a conclusion. Use this whenever the evidence is missing, unreadable, ambiguous, or beyond what your read-only tools can reach.
|
|
@@ -37,4 +46,47 @@ yours, and reporting it honestly is strictly better than guessing.
|
|
|
37
46
|
**Never report `fail` for want of evidence.** "I could not find it" is
|
|
38
47
|
`undeterminable`, not `fail`. Collapsing the two turns a tooling gap into an
|
|
39
48
|
accusation of incorrect work, and the runner acts on that accusation — it rolls
|
|
40
|
-
the covered tasks back and reopens work that may be perfectly fine.
|
|
49
|
+
the covered tasks back and reopens work that may be perfectly fine.
|
|
50
|
+
|
|
51
|
+
### 2. Implementation findings
|
|
52
|
+
|
|
53
|
+
Report defects anywhere in the implemented work — not only what the checks
|
|
54
|
+
cover. Scope is the whole change the plan drove, judged against the plan's
|
|
55
|
+
intent. If you find nothing worth reporting, emit exactly `- none.` under the
|
|
56
|
+
heading and stop.
|
|
57
|
+
|
|
58
|
+
One bullet per finding, exact line grammar (field order is fixed; fields are
|
|
59
|
+
separated by `; `):
|
|
60
|
+
|
|
61
|
+
- `F-###` — severity: high | medium | low; tasks: Task-N, Task-M | none; proposed-task: <imperative one-line title>; note: <one line>; evidence: <repo path or command output proving it>
|
|
62
|
+
|
|
63
|
+
Rules for the fields:
|
|
64
|
+
|
|
65
|
+
- `F-###` — a stable id. The brief lists the previous round's unresolved
|
|
66
|
+
findings: when a listed problem is still present, **reuse its id verbatim**
|
|
67
|
+
and do not renumber. A problem is resolved only by no longer reporting it.
|
|
68
|
+
Brand-new problems take the next unused number after the highest id you have
|
|
69
|
+
seen (in the brief or in this report).
|
|
70
|
+
- `severity` — `high` blocks completion and wakes the executor for a fix
|
|
71
|
+
round; `medium` and `low` are recorded and summarized at completion. Grade
|
|
72
|
+
by consequence: `high` = correctness, data loss, security, broken promised
|
|
73
|
+
behavior, or a verification check that is demonstrably unmet. `medium` =
|
|
74
|
+
should be fixed, but the plan's promised behavior still holds without it.
|
|
75
|
+
`low` = polish, naming, comments, minor drift. Do not inflate; do not
|
|
76
|
+
downgrade a real `high` to avoid waking the executor.
|
|
77
|
+
- `tasks` — the task id(s) whose work is defective, comma-separated, or
|
|
78
|
+
`none` when no existing task owns the defect. Only use ids from the brief's
|
|
79
|
+
task list. A `high` finding with `tasks: none` must carry a
|
|
80
|
+
`proposed-task:` field (below); the runner appends that task to the plan
|
|
81
|
+
mechanically, so write it as a self-contained imperative title (e.g.
|
|
82
|
+
`cap retry backoff at 60s in src/client.ts`). Omit `proposed-task:` for
|
|
83
|
+
mapped findings and for `medium`/`low`.
|
|
84
|
+
- `note` — one line: what is wrong and what breaks.
|
|
85
|
+
- `evidence` — a repository path (with line anchor when useful) or a short
|
|
86
|
+
quoted excerpt that an executor with read tools can re-inspect. A source
|
|
87
|
+
path plus a defect argument is sufficient; you have no execution tools, so
|
|
88
|
+
never fabricate command output.
|
|
89
|
+
|
|
90
|
+
Every reported defect must have a bullet. Never fold two defects into one
|
|
91
|
+
bullet; never mention a defect in prose without a bullet — findings outside
|
|
92
|
+
the grammar are invisible to the runner.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-plans",
|
|
3
|
-
"version": "0.8.
|
|
3
|
+
"version": "0.8.1",
|
|
4
4
|
"description": "Human-in-the-loop planning extension for the Pi coding agent: researched, refined Markdown plans before any code changes.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -41,7 +41,7 @@
|
|
|
41
41
|
"README.md",
|
|
42
42
|
"LICENSE",
|
|
43
43
|
"CONTRIBUTING.md",
|
|
44
|
-
|
|
44
|
+
"AGENTS.md",
|
|
45
45
|
"index.ts",
|
|
46
46
|
"agents/",
|
|
47
47
|
"docs/assets",
|
|
@@ -131,14 +131,15 @@ When the user picks `✓ Accept PLAN_vN and execute it now` in the merged questi
|
|
|
131
131
|
|
|
132
132
|
- every agent turn is injected with the current wave's open tasks, the remaining task list, verification-check summary, and execution rules (wave order, `plans_update_task` reporting with status + evidence / skipReason, subprocess polling backoff 5s -> 10s -> 20s -> 40s -> 80s then keep polling at 80s, no stopgaps, dependency and library discipline, minimum tests);
|
|
133
133
|
- task progress flows exclusively through the `plans_update_task` tool: one call per task closing it as `complete` (with evidence) or `skipped` (with skipReason); closed statuses are immutable outside the audit-authorized rollback channel; subtasks close before their parent;
|
|
134
|
-
- the task dashboard tracks the whole tree live: a compact aboveEditor widget (current task ▸, progress bar, ✓/· counts, VC pass count, wave indicator, pause state, audit-outcome line) and the Ctrl+Shift+T expanded tree view (✓/▸/~/·/↺ markers, current-task anchor, VC list with audit state, width-adaptive layout). `↺` marks a task that was completed and then rolled back by a failed audit: it is open again but still shows the evidence from its previous attempt;
|
|
134
|
+
- the task dashboard tracks the whole tree live: a compact aboveEditor widget (current task ▸, progress bar, ✓/· counts, VC pass count, wave indicator, pause state, audit-outcome line, unresolved-findings line visible in both the repairing and verifying phases) and the Ctrl+Shift+T expanded tree view (✓/▸/~/·/↺ markers, current-task anchor, VC list with audit state, width-adaptive layout). `↺` marks a task that was completed and then rolled back by a failed audit: it is open again but still shows the evidence from its previous attempt;
|
|
135
135
|
- a stall watchdog pauses execution after three consecutive settled rounds without any task-status change (genuine user input or `/plans-execute` resumes without losing progress). A round in which the agent ran successful tool calls counts as progress even with no status change, so investigating the codebase is never mistaken for a dead agent;
|
|
136
136
|
- when every task reaches a terminal state, the run status moves to `verifying` and an independent read-only execution reviewer (`agents/execution-reviewer.md`) verifies each check against the worktree in a detached round whose progress renders in a dedicated overlay (Esc closes it; Ctrl+Shift+R reopens the in-flight round). Each check gets one of three verdicts:
|
|
137
137
|
- `pass` — the evidence establishes the check's condition;
|
|
138
138
|
- `fail` — the condition is demonstrably not met; the covered tasks roll back to pending (children cascade, skipped tasks reopen, the `evidence` of the previous attempt is retained while `skipReason` is cleared) and the audit report is injected;
|
|
139
|
-
- `undeterminable` — the reviewer could not reach a conclusion. This is never a failure and never a completion: nothing rolls back, the loop self-schedules the retry (no agent wake), and each retry counts toward the five-round budget.
|
|
139
|
+
- `undeterminable` — the reviewer could not reach a conclusion. This is never a failure and never a completion: nothing rolls back, the loop self-schedules the retry (no agent wake), and each retry counts toward the five-round budget. An all-undeterminable round that also reports a high-severity finding still wakes the executor — a finding is actionable independent of verdict evidence;
|
|
140
|
+
- `F-###` findings — every round also reports severity-graded implementation findings (`high | medium | low`) over the whole implemented change, not just what the checks cover. Ids are stable: the brief lists the previous round's unresolved findings and the reviewer reuses their ids verbatim while the problem persists; absence from the newest round's report is the resolution signal. A `high` finding (or a `fail` verdict) opens ONE union fix round: mapped tasks roll back to pending exactly like a failed check's coverage (children cascade, evidence retained), and a `high` whose `tasks: none` names no owner gets a plan task appended mechanically from the reviewer's `proposed-task` title — the reviewer itself stays read-only. The executor is woken exactly once with a findings summary plus the round-report path. A pure finding-driven rollback deliberately keeps earlier VC passes (the findings channel re-examines the repaired work next round); only a `fail`-driven rollback invalidates the checks covering the reopened tasks. Unresolved findings persist across checkpoint restores, session restores, and fresh budget grants.
|
|
140
141
|
|
|
141
|
-
The run completes only when every pending check is affirmatively `pass
|
|
142
|
+
The run completes only when every pending check is affirmatively `pass` AND the newest round reports no high-severity finding; residual `medium`/`low` findings are summarized in the completion message and stay recorded in the round reports. A partial round credits the checks that passed and leaves the rest for the next round. Checks whose covered tasks are all skipped pass as skipped-pass; checks covering no task never enter the audit; checks already satisfied in an earlier round are neither re-briefed nor re-judged;
|
|
142
143
|
- the round budget is five COMMITTED rounds (pass, fail, or undeterminable; discarded fingerprint-mismatch attempts and cancellations burn nothing; two consecutive discards commit as one undeterminable round). Exhaustion pauses the run in EVERY mode — interactive and auto-approve/headless alike — with an in-band `pi-plans-review-paused` message; ordinary user input and session restores never lift the pause or refill the budget. The only fresh-budget surface is `/plans-execute`, whose explicit confirmation grants five more rounds. The watchdog counter is rebased by real tool activity;
|
|
143
144
|
- execution-phase compaction is handled only when Pi core emits manual `/compact`, threshold, or overflow events; summaries are deterministic VCC-style summaries, include session-derived plan/current-task/checklist context, use smart tail keep and `keep:N`, and never call a model;
|
|
144
145
|
- the read-only guard lifts: full write access returns;
|
|
@@ -203,11 +203,11 @@ One run directory per planning request: `<git-common-dir>/pi-plans/runs/<YYYYMMD
|
|
|
203
203
|
|
|
204
204
|
## Workflow Checkpoints (`/resume-plans`)
|
|
205
205
|
|
|
206
|
-
Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | completed`; the legacy 0.6.0 `implementation-review` phase is read-tolerated and maps to done), `nextAction`, the exact plan identity (path + version + SHA-256), pending/answered questions (stable `questionId`), review rounds with per-lane status and result-file references, execution approval evidence (plan digest, worktree, `git revparse HEAD` at approval, task progress map, audit rounds with the failed and undeterminable check sets, the watchdog budget counter, verified VC set, usage), and ownership metadata. Full review outputs live in separate `reviews/` files; the checkpoint keeps only validated references.
|
|
206
|
+
Each run may carry a `checkpoint.json` — the durable, cross-session workflow state that `/resume-plans` restores in the current session. It records: logical `phase` (`planning | reviewing | executing | completed`; the legacy 0.6.0 `implementation-review` phase is read-tolerated and maps to done), `nextAction`, the exact plan identity (path + version + SHA-256), pending/answered questions (stable `questionId`), review rounds with per-lane status and result-file references, execution approval evidence (plan digest, worktree, `git revparse HEAD` at approval, task progress map, audit rounds with the failed and undeterminable check sets plus the unresolved findings of the newest committed round, the watchdog budget counter, verified VC set, usage), and ownership metadata. Full review outputs live in separate `reviews/` files; the checkpoint keeps only validated references.
|
|
207
207
|
|
|
208
208
|
Rules:
|
|
209
209
|
|
|
210
|
-
- Validation is explicit: unknown schema versions, malformed shapes, and unexpected keys are rejected; missing and corrupt checkpoints are distinct, and corrupt files are never silently overwritten. Keys added by a later version (`execution.stallRounds`, `execution.audit.undeterminable`) are optional on read, so checkpoints written before them keep loading.
|
|
210
|
+
- Validation is explicit: unknown schema versions, malformed shapes, and unexpected keys are rejected; missing and corrupt checkpoints are distinct, and corrupt files are never silently overwritten. Keys added by a later version (`execution.stallRounds`, `execution.audit.undeterminable`, `execution.audit.findings`, `execution.planAmended`) are optional on read, so checkpoints written before them keep loading. `execution.planAmended` is set when the review loop mechanically appended finding tasks to the approved plan; the checkpoint's plan identity was re-stamped to the amended digest at that moment while the approval record keeps the original.
|
|
211
211
|
- Writes are atomic with monotonic revisions; writers may require ownership (token + generation) or an expected revision.
|
|
212
212
|
- Model-driven boundaries (plan written, review consolidated, termination condition recorded, implementation round finished, completed) go through the whitelisted `plans record-checkpoint` action, which enforces state-machine preconditions — it cannot set execution approval, mark VCs passed, or forge terminal states.
|
|
213
213
|
- `ask_choice` accepts `questionId`/`purpose`; a pending question is durable before the panel opens and the answer before it returns. When a crash leaves a question both answered (ledger) and pending (checkpoint), the answered entry wins.
|
package/src/auditor.ts
CHANGED
|
@@ -12,8 +12,8 @@
|
|
|
12
12
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
13
13
|
import * as fs from "node:fs";
|
|
14
14
|
import * as path from "node:path";
|
|
15
|
-
import { auditableChecks, skippedPassCheckIds, type TaskView } from "./tasks.ts";
|
|
16
|
-
import type
|
|
15
|
+
import { auditableChecks, flattenTaskViews, skippedPassCheckIds, type TaskView } from "./tasks.ts";
|
|
16
|
+
import { normalizeTaskId, type CheckItem } from "./plan.ts";
|
|
17
17
|
import { messaging } from "./messaging.ts";
|
|
18
18
|
|
|
19
19
|
/** Budget cap: committed rounds per user-granted budget. Discarded (fingerprint-changed) attempts do not count. */
|
|
@@ -29,9 +29,36 @@ export interface AuditOutcome {
|
|
|
29
29
|
/** Checks whose report carried no readable verdict, or whose evidence was
|
|
30
30
|
* inconclusive. Never treated as `failed`: see src/exec.ts. */
|
|
31
31
|
undeterminable: string[];
|
|
32
|
+
/** Severity-graded implementation findings from this round (v0.9). Absent
|
|
33
|
+
* on legacy shapes means "no findings reported"; the parse boundary always
|
|
34
|
+
* sets a concrete array so hand-written construction sites cannot silently
|
|
35
|
+
* drop them. */
|
|
36
|
+
findings?: ReviewFinding[];
|
|
32
37
|
report: string;
|
|
33
38
|
}
|
|
34
39
|
|
|
40
|
+
/** Severity vocabulary for implementation findings; mirrors the plan
|
|
41
|
+
* priority words. `malformed` marks a bullet the grammar parser could not
|
|
42
|
+
* read — it is recorded and displayed but never drives a rollback. */
|
|
43
|
+
export type FindingSeverity = "high" | "medium" | "low" | "malformed";
|
|
44
|
+
|
|
45
|
+
export interface ReviewFinding {
|
|
46
|
+
/** Stable id, normalized uppercase (`F-001`). Reused verbatim across
|
|
47
|
+
* rounds while the problem persists; absence from the newest round's
|
|
48
|
+
* report is the resolution signal. */
|
|
49
|
+
id: string;
|
|
50
|
+
severity: FindingSeverity;
|
|
51
|
+
/** Task ids owning the defect (normalized), `[]` when unmapped. */
|
|
52
|
+
taskIds: string[];
|
|
53
|
+
/** Required for unmapped high findings: the runner appends this as a new
|
|
54
|
+
* plan task mechanically, so it must be a self-contained imperative title. */
|
|
55
|
+
proposedTask?: string;
|
|
56
|
+
note: string;
|
|
57
|
+
evidence: string;
|
|
58
|
+
/** Original bullet, for degraded records and round reports. */
|
|
59
|
+
raw: string;
|
|
60
|
+
}
|
|
61
|
+
|
|
35
62
|
/** Parse the verdict lines of an audit report against the checks that are
|
|
36
63
|
* still pending. A pending check with no readable verdict is
|
|
37
64
|
* `undeterminable`, never `failed` — the caller must be able to tell "the
|
|
@@ -40,6 +67,58 @@ export interface ParsedAudit {
|
|
|
40
67
|
passed: string[];
|
|
41
68
|
failed: string[];
|
|
42
69
|
undeterminable: string[];
|
|
70
|
+
findings: ReviewFinding[];
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** One finding bullet's field-capture helper: lazily up to the next
|
|
74
|
+
* `; <known-field>:` boundary or the end of the line, so free text in one
|
|
75
|
+
* field cannot swallow the next. */
|
|
76
|
+
function captureField(line: string, field: string): string | undefined {
|
|
77
|
+
const m = line.match(new RegExp(`${field}:\\s*(.*?)(?=;\\s*(?:severity|tasks|proposed-task|note|evidence):|$)`, "i"));
|
|
78
|
+
return m ? m[1].trim().replace(/^[*_`~]+|[*_`~]+$/g, "") : undefined;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** Parse the findings bullets of a review report. Degrade, never crash, never
|
|
82
|
+
* roll back: a bullet with an F-### id but unreadable fields is recorded with
|
|
83
|
+
* severity "malformed" (visible, non-blocking), exactly like the verdict
|
|
84
|
+
* parser's emphasis tolerance — a strict grammar once silently discarded
|
|
85
|
+
* verdicts and fail-closed rolled correct work back. */
|
|
86
|
+
export function parseFindings(report: string, knownTaskIds?: Set<string>): ReviewFinding[] {
|
|
87
|
+
const findings: ReviewFinding[] = [];
|
|
88
|
+
const seen = new Set<string>();
|
|
89
|
+
for (const match of report.matchAll(/^\s*[-*]\s+`?(F-\d+)`?\b/gim)) {
|
|
90
|
+
const id = match[1].toUpperCase();
|
|
91
|
+
if (seen.has(id)) continue; // conflicting duplicates resolve to the first
|
|
92
|
+
seen.add(id);
|
|
93
|
+
// The bullet regex's leading \s* may swallow the preceding newline, so
|
|
94
|
+
// slice(index) can start mid-whitespace; strip it before taking the line.
|
|
95
|
+
const line = match.input!.slice(match.index!).replace(/^[\s]+/, "").split(/\n/)[0];
|
|
96
|
+
const severityRaw = captureField(line, "severity")?.toLowerCase();
|
|
97
|
+
const tasksRaw = captureField(line, "tasks");
|
|
98
|
+
const taskIds = (tasksRaw ?? "")
|
|
99
|
+
.split(/[,,,、]/)
|
|
100
|
+
.map((token) => normalizeTaskId(token.trim()))
|
|
101
|
+
.filter((t): t is string => t !== null)
|
|
102
|
+
.filter((t) => !knownTaskIds || knownTaskIds.has(t));
|
|
103
|
+
const severity: FindingSeverity =
|
|
104
|
+
severityRaw === "high" || severityRaw === "medium" || severityRaw === "low" ? severityRaw : "malformed";
|
|
105
|
+
if (severity === "malformed" || !tasksRaw) {
|
|
106
|
+
// Unreadable severity, or the grammar's mandatory tasks field missing:
|
|
107
|
+
// degrade to a recorded non-blocking entry.
|
|
108
|
+
findings.push({ id, severity: "malformed", taskIds: [], note: captureField(line, "note") ?? line.trim(), evidence: captureField(line, "evidence") ?? "", raw: line.trim() });
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
findings.push({
|
|
112
|
+
id,
|
|
113
|
+
severity,
|
|
114
|
+
taskIds,
|
|
115
|
+
proposedTask: captureField(line, "proposed-task") || undefined,
|
|
116
|
+
note: captureField(line, "note") ?? "",
|
|
117
|
+
evidence: captureField(line, "evidence") ?? "",
|
|
118
|
+
raw: line.trim(),
|
|
119
|
+
});
|
|
120
|
+
}
|
|
121
|
+
return findings;
|
|
43
122
|
}
|
|
44
123
|
|
|
45
124
|
/** Checks this round must judge: auditable (covering at least one task) and
|
|
@@ -51,13 +130,27 @@ export function auditablePendingChecks(checklist: CheckItem[], tasks: TaskView[]
|
|
|
51
130
|
}
|
|
52
131
|
|
|
53
132
|
/** Build the audit brief for the read-only subagent. Exported for tests. */
|
|
54
|
-
export function buildAuditTask(
|
|
133
|
+
export function buildAuditTask(
|
|
134
|
+
planPath: string,
|
|
135
|
+
checklist: CheckItem[],
|
|
136
|
+
tasks: TaskView[],
|
|
137
|
+
round: number,
|
|
138
|
+
priorFindings: ReviewFinding[] = [],
|
|
139
|
+
): string {
|
|
55
140
|
const checks = auditablePendingChecks(checklist, tasks)
|
|
56
141
|
.map((check) => `- \`${check.id}\`: ${check.text}`)
|
|
57
142
|
.join("\n");
|
|
58
|
-
|
|
143
|
+
const taskList = flattenTaskViews(tasks)
|
|
144
|
+
.map((t) => `- \`${t.id}\`: ${t.title} (${t.status})`)
|
|
145
|
+
.join("\n");
|
|
146
|
+
const prior = priorFindings.length
|
|
147
|
+
? `Unresolved findings from earlier rounds (reuse these exact ids while the problem persists; a problem is resolved only by no longer reporting it):
|
|
59
148
|
|
|
60
|
-
|
|
149
|
+
${priorFindings.map((f) => `- \`${f.id}\` — severity: ${f.severity}; tasks: ${f.taskIds.join(", ") || "none"}; note: ${f.note}`).join("\n")}`
|
|
150
|
+
: "(none — this is the first round with findings in scope)";
|
|
151
|
+
return `Goal: verify that the implemented worktree satisfies the accepted plan's verification checks, and report implementation findings that drive the fix loop.
|
|
152
|
+
|
|
153
|
+
Target plan: ${planPath} (review round ${round})
|
|
61
154
|
|
|
62
155
|
Authority boundary: read-only analysis only. Do not edit, write, delete, commit, push, or spawn subagents.
|
|
63
156
|
|
|
@@ -67,36 +160,75 @@ Checks to verify (only these; checks covering no task and checks already satisfi
|
|
|
67
160
|
|
|
68
161
|
${checks}
|
|
69
162
|
|
|
70
|
-
|
|
163
|
+
Plan tasks (the only ids valid in a finding's tasks field):
|
|
164
|
+
|
|
165
|
+
${taskList}
|
|
166
|
+
|
|
167
|
+
${prior}
|
|
168
|
+
|
|
169
|
+
Output: exactly two sections, in this order.
|
|
170
|
+
|
|
171
|
+
1. Verification verdicts — one section per check, in the order above:
|
|
71
172
|
|
|
72
173
|
- \`VC-###\` — verdict: pass | fail | undeterminable; evidence: <repo path/command proving it>; note: <one line>.
|
|
73
174
|
|
|
74
|
-
Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence
|
|
175
|
+
Emit every listed check exactly once. \`undeterminable\` is a legitimate answer: use it whenever the evidence is missing, unreadable, ambiguous, or beyond your read-only reach. Report \`fail\` only when you can point at the specific thing that breaks the condition — never report \`fail\` for want of evidence.
|
|
176
|
+
|
|
177
|
+
2. Implementation findings — one bullet per defect anywhere in the implemented change (not only what the checks cover), exact line grammar:
|
|
178
|
+
|
|
179
|
+
- \`F-###\` — severity: high | medium | low; tasks: Task-N, Task-M | none; proposed-task: <imperative one-line title>; note: <one line>; evidence: <repo path or quoted excerpt>
|
|
180
|
+
|
|
181
|
+
\`high\` wakes the executor for a fix round; \`medium\`/\`low\` are recorded. When no existing task owns the defect use \`tasks: none\` and (required for high) a self-contained \`proposed-task:\` title. If nothing is worth reporting, emit exactly \`- none.\` under the findings heading.`;
|
|
75
182
|
}
|
|
76
183
|
|
|
77
184
|
/** Parse the audit subagent's verdict lines. Exported for tests. */
|
|
78
|
-
export function parseAuditReport(report: string, pendingVcIds: string[]): ParsedAudit {
|
|
185
|
+
export function parseAuditReport(report: string, pendingVcIds: string[], knownTaskIds?: Set<string>): ParsedAudit {
|
|
79
186
|
const passed: string[] = [];
|
|
80
187
|
const failed: string[] = [];
|
|
81
188
|
const known = new Set(pendingVcIds.map((id) => id.toUpperCase()));
|
|
82
|
-
// The
|
|
189
|
+
// The reviewer writes Markdown, so a verdict may carry emphasis
|
|
83
190
|
// (`**pass**`, `*pass*`, `_pass_`, `` `pass` ``). Requiring a bare token
|
|
84
191
|
// silently discarded such verdicts, and the caller's fail-closed rule then
|
|
85
192
|
// marked every check failed and rolled the whole run back -- reporting
|
|
86
193
|
// correct work as failure. Tolerate the markers; \b keeps `passed` and
|
|
87
194
|
// `passing` from matching.
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
195
|
+
// v0.9: finding bullets (F-###) are excluded from verdict scanning — a
|
|
196
|
+
// finding's note may cite a VC id, and that must never register a verdict.
|
|
197
|
+
// v0.9.1 (F-011): the contract promises "one section per check" and an
|
|
198
|
+
// equally literal reading puts the id in a `### VC-###` heading with the
|
|
199
|
+
// verdict on a line of its own below it. The old same-line-only scan
|
|
200
|
+
// parsed such reports to ZERO verdicts and burned whole budgets as
|
|
201
|
+
// undeterminable. The scan is now section-aware: a line that NAMES a
|
|
202
|
+
// known check at a heading/bullet start opens that check's section, and a
|
|
203
|
+
// bare `verdict:` token attributes to the nearest open section; an
|
|
204
|
+
// id-and-verdict pair on one line stays direct.
|
|
205
|
+
const record = (id: string, verdict: string): void => {
|
|
92
206
|
if (verdict === "pass") passed.push(id);
|
|
93
207
|
else if (verdict === "fail") failed.push(id);
|
|
208
|
+
};
|
|
209
|
+
const verdictToken = /verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i;
|
|
210
|
+
let current: string | null = null;
|
|
211
|
+
for (const line of report.split(/\n/)) {
|
|
212
|
+
if (/^\s*[-*]\s+`?F-\d+`?\b/i.test(line)) continue; // finding bullet
|
|
213
|
+
const sectionId = line.match(/^\s*(?:#{1,6}\s+|[-*]\s+)?`?(VC-\d+)`?\b/i);
|
|
214
|
+
if (sectionId) {
|
|
215
|
+
const id = sectionId[1].toUpperCase();
|
|
216
|
+
if (known.has(id)) current = id;
|
|
217
|
+
}
|
|
218
|
+
const direct = line.match(/`?(VC-\d+)`?[^\n]*?verdict:\s*[*_`~]*\s*(pass|fail|undeterminable)\b/i);
|
|
219
|
+
if (direct) {
|
|
220
|
+
const id = direct[1].toUpperCase();
|
|
221
|
+
if (known.has(id)) record(id, direct[2].toLowerCase());
|
|
222
|
+
continue;
|
|
223
|
+
}
|
|
224
|
+
const bare = line.match(verdictToken);
|
|
225
|
+
if (bare && current) record(current, bare[1].toLowerCase());
|
|
94
226
|
}
|
|
95
227
|
// A check with conflicting verdicts resolves to fail: the reader saw both.
|
|
96
228
|
for (const id of [...new Set(passed)]) if (failed.includes(id)) passed.splice(passed.indexOf(id), 1);
|
|
97
229
|
const undecided = new Set(known);
|
|
98
230
|
for (const id of [...passed, ...failed]) undecided.delete(id);
|
|
99
|
-
return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided] };
|
|
231
|
+
return { passed: [...new Set(passed)], failed: [...new Set(failed)], undeterminable: [...undecided], findings: parseFindings(report, knownTaskIds) };
|
|
100
232
|
}
|
|
101
233
|
|
|
102
234
|
/** Pure decision core: classify a parsed report into the audit outcome. It
|
|
@@ -109,6 +241,7 @@ export function applyAuditOutcome(round: number, parsed: ParsedAudit, report: st
|
|
|
109
241
|
passed: parsed.passed,
|
|
110
242
|
failed: parsed.failed,
|
|
111
243
|
undeterminable: parsed.undeterminable,
|
|
244
|
+
findings: parsed.findings,
|
|
112
245
|
report,
|
|
113
246
|
};
|
|
114
247
|
}
|
|
@@ -132,6 +265,9 @@ export async function runCompletionAudit(
|
|
|
132
265
|
checklist: CheckItem[];
|
|
133
266
|
tasks: TaskView[];
|
|
134
267
|
round: number;
|
|
268
|
+
/** Unresolved findings from earlier committed rounds, injected into
|
|
269
|
+
* the brief so the reviewer reuses stable ids (v0.9). */
|
|
270
|
+
priorFindings?: ReviewFinding[];
|
|
135
271
|
model?: string;
|
|
136
272
|
thinkingLevel?: string;
|
|
137
273
|
timeoutMs?: number;
|
|
@@ -141,7 +277,7 @@ export async function runCompletionAudit(
|
|
|
141
277
|
): Promise<AuditRoundResult> {
|
|
142
278
|
const { runPiSubagent } = await import("./subagent.ts");
|
|
143
279
|
const pending = auditablePendingChecks(opts.checklist, opts.tasks);
|
|
144
|
-
const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round);
|
|
280
|
+
const task = buildAuditTask(opts.planPath, opts.checklist, opts.tasks, opts.round, opts.priorFindings ?? []);
|
|
145
281
|
// agents/auditor.md, not agents/reviewer.md: the reviewer prompt mandates a
|
|
146
282
|
// plan-review shape (`## Findings` / `## Questions`, F-###) and never says
|
|
147
283
|
// "verdict", so auditing under it produced reports this parser could not
|
|
@@ -165,13 +301,14 @@ export async function runCompletionAudit(
|
|
|
165
301
|
if (result.cancelled === true) return { cancelled: true };
|
|
166
302
|
return null;
|
|
167
303
|
}
|
|
168
|
-
const parsed = parseAuditReport(result.output, pending.map((item) => item.id));
|
|
304
|
+
const parsed = parseAuditReport(result.output, pending.map((item) => item.id), new Set(flattenTaskViews(opts.tasks).map((t) => t.id)));
|
|
169
305
|
messaging().appendEntry("pi-plans-audit", {
|
|
170
306
|
planPath: opts.planPath,
|
|
171
307
|
round: opts.round,
|
|
172
308
|
passed: parsed.passed,
|
|
173
309
|
failed: parsed.failed,
|
|
174
310
|
undeterminable: parsed.undeterminable,
|
|
311
|
+
highFindings: parsed.findings.filter((f) => f.severity === "high").map((f) => f.id),
|
|
175
312
|
});
|
|
176
313
|
return applyAuditOutcome(opts.round, parsed, result.output);
|
|
177
314
|
}
|
|
@@ -191,6 +328,9 @@ export function writeReviewRoundReport(
|
|
|
191
328
|
passed: string[];
|
|
192
329
|
failed: string[];
|
|
193
330
|
undeterminable: string[];
|
|
331
|
+
/** v0.9 findings from this round; recorded in the round report so the
|
|
332
|
+
* file stays the full evidence record. */
|
|
333
|
+
findings?: ReviewFinding[];
|
|
194
334
|
discardedReason?: string;
|
|
195
335
|
fingerprintCaptured?: string;
|
|
196
336
|
fingerprintFound?: string;
|
|
@@ -209,6 +349,8 @@ export function writeReviewRoundReport(
|
|
|
209
349
|
`- passed: ${entry.passed.join(", ") || "(none)"}`,
|
|
210
350
|
`- failed: ${entry.failed.join(", ") || "(none)"}`,
|
|
211
351
|
`- undeterminable: ${entry.undeterminable.join(", ") || "(none)"}`,
|
|
352
|
+
`- high findings: ${entry.findings?.filter((f) => f.severity === "high").map((f) => f.id).join(", ") || "(none)"}`,
|
|
353
|
+
...(entry.findings?.length ? [`- findings: ${entry.findings.map((f) => `${f.id} (${f.severity}${f.proposedTask ? "; proposed: " + f.proposedTask : ""})`).join(" | ")}`] : []),
|
|
212
354
|
...(entry.discardedReason ? [`- discarded: ${entry.discardedReason}`] : []),
|
|
213
355
|
`- fingerprint (captured): ${entry.fingerprintCaptured ?? "(n/a)"}`,
|
|
214
356
|
`- fingerprint (at resolve): ${entry.fingerprintFound ?? "(n/a)"}`,
|
package/src/dashboard.ts
CHANGED
|
@@ -46,6 +46,9 @@ export interface DashboardModel {
|
|
|
46
46
|
* never render `audit complete ✓` while this is set — the round-1 mis-cue
|
|
47
47
|
* showed the tick for the whole duration of a running audit. */
|
|
48
48
|
reviewRunning: boolean;
|
|
49
|
+
/** v0.9: unresolved findings from the newest committed review round
|
|
50
|
+
* (stable ids). High entries block completion; the rest are recorded. */
|
|
51
|
+
findings: Array<{ id: string; severity: string; note: string; taskIds: string[] }>;
|
|
49
52
|
startedAt: string;
|
|
50
53
|
usage: { inToks: number; outToks: number };
|
|
51
54
|
}
|
|
@@ -172,22 +175,33 @@ export function renderDashboardLines(model: DashboardModel, width: number, theme
|
|
|
172
175
|
// instead — the panel must look alive, never finished-then-silent.
|
|
173
176
|
const owed = model.checklist.some((item) => !item.done);
|
|
174
177
|
const nextRound = (model.auditRounds ?? 0) + 1;
|
|
178
|
+
const highCount = model.findings.filter((f) => f.severity === "high").length;
|
|
175
179
|
const auditLine = model.reviewRunning
|
|
176
180
|
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} running — read-only reviewer verifying`
|
|
177
181
|
: model.auditFailed.length > 0
|
|
178
182
|
? `audit: ${model.auditFailed.length} check(s) failed — rollback pending`
|
|
179
|
-
:
|
|
180
|
-
? `
|
|
181
|
-
:
|
|
182
|
-
? `
|
|
183
|
-
:
|
|
184
|
-
?
|
|
185
|
-
:
|
|
183
|
+
: highCount > 0
|
|
184
|
+
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — ${highCount} high finding(s) unresolved`
|
|
185
|
+
: model.auditUndeterminable.length > 0
|
|
186
|
+
? `audit: ${model.auditUndeterminable.length} check(s) undeterminable — verdict unreadable`
|
|
187
|
+
: owed
|
|
188
|
+
? `review: round ${nextRound}/${REVIEW_MAX_ROUNDS} — verdict pending`
|
|
189
|
+
: model.auditRounds !== null
|
|
190
|
+
? "audit complete ✓"
|
|
191
|
+
: "all tasks terminal — audit pending";
|
|
186
192
|
lines.push(boxRow("│", ` ${clip(auditLine, inner - 2)}`, " ", width));
|
|
187
193
|
}
|
|
188
194
|
if (!narrow && model.auditFailed.length > 0) {
|
|
189
195
|
lines.push(boxRow("│", ` ✗ ${clip(model.auditFailed.join(", "), inner - 3)}`, " ", width));
|
|
190
196
|
}
|
|
197
|
+
// v0.9: unresolved findings render in both phases — the fix loop's whole
|
|
198
|
+
// point is that the executor sees what it owes while repairing.
|
|
199
|
+
const highFindings = model.findings.filter((f) => f.severity === "high");
|
|
200
|
+
if (!narrow && highFindings.length > 0) {
|
|
201
|
+
lines.push(boxRow("│", ` ⚠ ${clip(`high: ${highFindings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
|
|
202
|
+
} else if (!narrow && model.findings.length > 0) {
|
|
203
|
+
lines.push(boxRow("│", ` · ${clip(`findings: ${model.findings.map((f) => f.id).join(", ")}`, inner - 3)}`, " ", width));
|
|
204
|
+
}
|
|
191
205
|
if (!narrow && model.auditUndeterminable.length > 0) {
|
|
192
206
|
lines.push(boxRow("│", ` ? ${clip(model.auditUndeterminable.join(", "), inner - 3)}`, " ", width));
|
|
193
207
|
}
|
|
@@ -216,7 +230,7 @@ export function deriveDashboardModel(
|
|
|
216
230
|
topic: string,
|
|
217
231
|
tasks: TaskView[],
|
|
218
232
|
checklist: CheckItem[],
|
|
219
|
-
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
233
|
+
extra?: { paused?: boolean; pausedReason?: string; auditRounds?: number | null; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean; findings?: Array<{ id: string; severity: string; note: string; taskIds: string[] }>; startedAt?: string; usage?: { inToks: number; outToks: number } },
|
|
220
234
|
): DashboardModel {
|
|
221
235
|
return {
|
|
222
236
|
topic,
|
|
@@ -228,6 +242,7 @@ export function deriveDashboardModel(
|
|
|
228
242
|
auditFailed: extra?.auditFailed ?? [],
|
|
229
243
|
auditUndeterminable: extra?.auditUndeterminable ?? [],
|
|
230
244
|
reviewRunning: extra?.reviewRunning ?? false,
|
|
245
|
+
findings: extra?.findings ?? [],
|
|
231
246
|
startedAt: extra?.startedAt ?? new Date().toISOString(),
|
|
232
247
|
usage: extra?.usage ?? { inToks: 0, outToks: 0 },
|
|
233
248
|
};
|
|
@@ -239,9 +254,14 @@ export function formatDashboardSummaryLine(model: DashboardModel): string {
|
|
|
239
254
|
const vcDone = model.checklist.filter((item) => item.done).length;
|
|
240
255
|
const cur = currentTask(model.tasks);
|
|
241
256
|
const wave = cur ? ` · wave ${cur.wave}` : "";
|
|
242
|
-
|
|
257
|
+
// v0.9: the review token carries the budget denominator and the unresolved
|
|
258
|
+
// high count — visible in BOTH phases (executing repair and verifying), so
|
|
259
|
+
// convergence is legible exactly while the executor is fixing.
|
|
260
|
+
const highs = model.findings.filter((f) => f.severity === "high").length;
|
|
261
|
+
const audit = model.auditRounds !== null ? ` · review r${model.auditRounds}/${REVIEW_MAX_ROUNDS}` : "";
|
|
262
|
+
const highToken = highs > 0 ? ` · ${highs} high` : "";
|
|
243
263
|
const pause = model.paused ? " · ⏸ paused" : "";
|
|
244
|
-
return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${pause}`;
|
|
264
|
+
return `plans: ${model.topic} ▸ tasks ${p.done}/${p.total} · VC ${vcDone}/${model.checklist.length}${wave}${audit}${highToken}${pause}`;
|
|
245
265
|
}
|
|
246
266
|
|
|
247
267
|
/** Expanded tree view lines (Ctrl+Shift+T overlay). Wide layout from 96 cols. */
|
|
@@ -266,23 +286,35 @@ export function renderDashboardTreeLines(model: DashboardModel, width: number, t
|
|
|
266
286
|
if (task.status === "pending" && task.evidence) line += ` ↺${clip(task.evidence, 40)}`;
|
|
267
287
|
lines.push(line);
|
|
268
288
|
for (const child of task.children) row(child, depth + 1);
|
|
269
|
-
};
|
|
289
|
+
};
|
|
290
|
+
for (const task of model.tasks) row(task, 0);
|
|
270
291
|
lines.push("");
|
|
271
292
|
lines.push("Verification checks:");
|
|
272
293
|
for (const item of model.checklist) {
|
|
273
294
|
const mark = model.auditFailed.includes(item.id) ? "✗" : item.done ? "☑" : "☐";
|
|
274
295
|
lines.push(` ${mark} ${item.id}${wide ? ` ${clip(item.text.split(";")[1] ?? item.text, Math.min(80, width))}` : ""}`);
|
|
275
296
|
}
|
|
297
|
+
if (model.findings.length > 0) {
|
|
298
|
+
lines.push("");
|
|
299
|
+
lines.push("Findings (stable ids; absence from the newest round = resolved):");
|
|
300
|
+
for (const f of model.findings) {
|
|
301
|
+
const mark = f.severity === "high" ? "⚠" : f.severity === "malformed" ? "?" : "·";
|
|
302
|
+
lines.push(` ${mark} ${f.id} (${f.severity}${f.taskIds.length ? `, ${f.taskIds.join(", ")}` : ""}): ${wide ? clip(f.note, 72) : clip(f.note, 40)}`);
|
|
303
|
+
}
|
|
304
|
+
}
|
|
276
305
|
if (model.reviewRunning) {
|
|
277
306
|
lines.push("");
|
|
278
307
|
lines.push(`Execution review: round ${(model.auditRounds ?? 0) + 1}/${REVIEW_MAX_ROUNDS} running`);
|
|
279
308
|
} else if (model.auditRounds !== null) {
|
|
280
309
|
lines.push("");
|
|
310
|
+
const highIds = model.findings.filter((f) => f.severity === "high").map((f) => f.id);
|
|
281
311
|
const verdict = model.auditFailed.length > 0
|
|
282
|
-
? ` — failed: ${model.auditFailed.join(", ")}`
|
|
283
|
-
:
|
|
284
|
-
? ` —
|
|
285
|
-
:
|
|
312
|
+
? ` — failed: ${model.auditFailed.join(", ")}${highIds.length > 0 ? `; high: ${highIds.join(", ")}` : ""}`
|
|
313
|
+
: highIds.length > 0
|
|
314
|
+
? ` — high findings unresolved: ${highIds.join(", ")}`
|
|
315
|
+
: model.auditUndeterminable.length > 0
|
|
316
|
+
? ` — undeterminable: ${model.auditUndeterminable.join(", ")}`
|
|
317
|
+
: " — passed ✓";
|
|
286
318
|
lines.push(`Execution review: round ${model.auditRounds}${verdict}`);
|
|
287
319
|
}
|
|
288
320
|
const painted = theme ? lines.map((line) => theme.fg("muted", line)) : lines;
|