pi-plans 0.8.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/agents/execution-reviewer.md +62 -10
- package/package.json +2 -2
- package/references/pi-planning-workflow.md +4 -3
- package/references/state-and-config.md +2 -2
- package/src/auditor.ts +158 -16
- package/src/dashboard.ts +47 -15
- package/src/exec.ts +282 -77
- package/src/refine-ui.ts +21 -5
- package/src/resume-command.ts +7 -1
- package/src/tasks.ts +23 -0
- package/src/workflow-state.ts +80 -6
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +185 -1
- package/tests/dashboard.test.ts +67 -1
- package/tests/exec-review-loop.test.ts +400 -7
- package/tests/refine-ui.test.ts +25 -2
- package/tests/workflow-state.test.ts +40 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/refine.ts +22 -3
package/src/refine-ui.ts
CHANGED
|
@@ -183,10 +183,15 @@ export class RefineOverlayComponent implements Component {
|
|
|
183
183
|
private readonly modelLabel?: string;
|
|
184
184
|
/** Chrome language (issue #3); defaults to English for direct construction. */
|
|
185
185
|
private readonly lang: UiLanguage;
|
|
186
|
+
/** v0.8.1: pi-tui routes input ONLY to the focused component (no
|
|
187
|
+
* bubbling), so an open overlay otherwise swallows every global shortcut
|
|
188
|
+
* — including the Ctrl+Shift+T dashboard toggle users expect to work while
|
|
189
|
+
* watching a review/refine overlay. Unhandled keys are forwarded here. */
|
|
190
|
+
private readonly onUnhandledKey?: (data: string) => void;
|
|
186
191
|
private selectedLane = 0;
|
|
187
192
|
private disposed = false;
|
|
188
193
|
|
|
189
|
-
constructor(theme: Theme, role: RefineOverlayRole, lanes: RefineLaneState[], onCancel: () => void, tui?: TUI, modelLabel?: string, lang: UiLanguage = "en") {
|
|
194
|
+
constructor(theme: Theme, role: RefineOverlayRole, lanes: RefineLaneState[], onCancel: () => void, tui?: TUI, modelLabel?: string, lang: UiLanguage = "en", onUnhandledKey?: (data: string) => void) {
|
|
190
195
|
this.theme = theme;
|
|
191
196
|
this.role = role;
|
|
192
197
|
this.lanes = lanes;
|
|
@@ -194,6 +199,7 @@ export class RefineOverlayComponent implements Component {
|
|
|
194
199
|
this.tui = tui;
|
|
195
200
|
this.modelLabel = modelLabel;
|
|
196
201
|
this.lang = lang;
|
|
202
|
+
this.onUnhandledKey = onUnhandledKey;
|
|
197
203
|
this.tui?.terminal?.write?.("\x1b[?1000h\x1b[?1006h");
|
|
198
204
|
}
|
|
199
205
|
|
|
@@ -214,7 +220,10 @@ export class RefineOverlayComponent implements Component {
|
|
|
214
220
|
return;
|
|
215
221
|
}
|
|
216
222
|
const lane = this.lanes[this.selectedLane];
|
|
217
|
-
if (!lane)
|
|
223
|
+
if (!lane) {
|
|
224
|
+
this.onUnhandledKey?.(data);
|
|
225
|
+
return;
|
|
226
|
+
}
|
|
218
227
|
const viewport = Math.max(1, lane.viewportHeight ?? 1);
|
|
219
228
|
if (matchesTerminalKey(data, "up")) lane.scrollOffset -= 1;
|
|
220
229
|
else if (matchesTerminalKey(data, "down")) lane.scrollOffset += 1;
|
|
@@ -222,7 +231,11 @@ export class RefineOverlayComponent implements Component {
|
|
|
222
231
|
else if (matchesTerminalKey(data, "pageDown")) lane.scrollOffset += Math.max(1, viewport - 1);
|
|
223
232
|
else {
|
|
224
233
|
const mouse = data.match(/^\x1b\[<(\d+);\d+;\d+[Mm]$/);
|
|
225
|
-
if (!mouse || (Number(mouse[1]) & 64) !== 64)
|
|
234
|
+
if (!mouse || (Number(mouse[1]) & 64) !== 64) {
|
|
235
|
+
// Not ours: forward instead of swallowing (see onUnhandledKey note).
|
|
236
|
+
this.onUnhandledKey?.(data);
|
|
237
|
+
return;
|
|
238
|
+
}
|
|
226
239
|
lane.scrollOffset += (Number(mouse[1]) & 1) === 0 ? -3 : 3;
|
|
227
240
|
}
|
|
228
241
|
lane.followTranscript = false;
|
|
@@ -303,11 +316,14 @@ export class RefineOverlayController {
|
|
|
303
316
|
private tui: TUI | undefined;
|
|
304
317
|
private closed = false;
|
|
305
318
|
private readonly lang: UiLanguage;
|
|
319
|
+
/** Forwarded unhandled keys (see RefineOverlayComponent.onUnhandledKey). */
|
|
320
|
+
private readonly onUnhandledKey?: (data: string) => void;
|
|
306
321
|
|
|
307
|
-
constructor(role: RefineOverlayRole, laneIds: Array<{ id: string; label?: string }>, onCancel: () => void, lang: UiLanguage = "en") {
|
|
322
|
+
constructor(role: RefineOverlayRole, laneIds: Array<{ id: string; label?: string }>, onCancel: () => void, lang: UiLanguage = "en", onUnhandledKey?: (data: string) => void) {
|
|
308
323
|
this.role = role;
|
|
309
324
|
this.onCancel = onCancel;
|
|
310
325
|
this.lang = lang;
|
|
326
|
+
this.onUnhandledKey = onUnhandledKey;
|
|
311
327
|
this.lanes = laneIds.map((lane) => ({
|
|
312
328
|
id: lane.id,
|
|
313
329
|
label: lane.label ?? lane.id,
|
|
@@ -330,7 +346,7 @@ export class RefineOverlayController {
|
|
|
330
346
|
(_tui, theme, _keybindings, done) => {
|
|
331
347
|
this.tui = _tui;
|
|
332
348
|
this.done = done;
|
|
333
|
-
this.component = new RefineOverlayComponent(theme, this.role, this.lanes, () => this.cancel(), _tui, modelLabel, this.lang);
|
|
349
|
+
this.component = new RefineOverlayComponent(theme, this.role, this.lanes, () => this.cancel(), _tui, modelLabel, this.lang, this.onUnhandledKey);
|
|
334
350
|
if (this.closed) done(undefined);
|
|
335
351
|
return this.component;
|
|
336
352
|
},
|
package/src/resume-command.ts
CHANGED
|
@@ -309,12 +309,18 @@ async function buildBrief(
|
|
|
309
309
|
: `\nExecution had been paused: ${load.pausedReason} — the pause is cleared by this resume; continue from where it stopped.`
|
|
310
310
|
: "";
|
|
311
311
|
const legacy = load.legacyPlan ? "\nThis plan parses through the legacy I-### compatibility mapping; upgrade it to the ## Tasks format at the next revision." : "";
|
|
312
|
+
// v0.9.1 (F-005): outstanding highs surface in the brief itself, not
|
|
313
|
+
// only in the per-turn injection.
|
|
314
|
+
const highs = (load.findings ?? []).filter((f) => f.severity === "high");
|
|
315
|
+
const highLine = highs.length > 0
|
|
316
|
+
? `\nUnresolved high-severity findings from review round (stable ids): ${highs.map((f) => `${f.id}${f.taskIds.length ? ` (${f.taskIds.join(", ")})` : ""}: ${f.note}`).join("; ")} — fix them, then re-close the affected tasks.`
|
|
317
|
+
: "";
|
|
312
318
|
// v0.8: a verifying run keeps checkpoint phase "executing" but the run
|
|
313
319
|
// STATUS is verifying — surface which loop owns the run right now.
|
|
314
320
|
const verifying = run.status === "verifying";
|
|
315
321
|
return {
|
|
316
322
|
phaseLabel: verifying ? "verifying" : "executing",
|
|
317
|
-
text: `[PI-PLANS RESUME] ${verifying ? "Execution review of" : "Execution of"} run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}\n${verifying ? "The task tree is terminal and the execution-review loop owns the run: when all tasks are terminal and checks are still owed, a read-only reviewer round runs automatically (status verifying → done when every check passes). If a check fails, its tasks roll back to pending — fix and re-close them with plans_update_task." : "Follow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the execution reviewer verify the checks. The current wave and remaining tasks are injected each turn."}`,
|
|
323
|
+
text: `[PI-PLANS RESUME] ${verifying ? "Execution review of" : "Execution of"} run ${runId} continues in this session.\nPlan: ${load.planPath}${reverify}${paused}${legacy}${highLine}\n${verifying ? "The task tree is terminal and the execution-review loop owns the run: when all tasks are terminal and checks are still owed, a read-only reviewer round runs automatically (status verifying → done when every check passes). If a check fails, its tasks roll back to pending — fix and re-close them with plans_update_task." : "Follow the execution-loop contract: work through tasks in wave order, report every task with the plans_update_task tool (status + evidence / skipReason), and let the execution reviewer verify the checks. The current wave and remaining tasks are injected each turn."}`,
|
|
318
324
|
};
|
|
319
325
|
}
|
|
320
326
|
if (load.legacyDelegate) {
|
package/src/tasks.ts
CHANGED
|
@@ -165,6 +165,29 @@ export function auditRollbackSet(
|
|
|
165
165
|
return reopen;
|
|
166
166
|
}
|
|
167
167
|
|
|
168
|
+
/** Rollback set for high-severity findings (v0.9): reopen exactly the named
|
|
169
|
+
* tasks — parents cascade to their children, skipped tasks reopen too — with
|
|
170
|
+
* semantics identical to auditRollbackSet, but keyed by task id because
|
|
171
|
+
* findings carry task ids, not VC ids. Evidence is kept and skipReason
|
|
172
|
+
* cleared for the same reasons as the VC path. */
|
|
173
|
+
export function findingsRollbackSet(tasks: TaskView[], taskIds: string[]): string[] {
|
|
174
|
+
const wanted = new Set(taskIds);
|
|
175
|
+
const flat = flattenTaskViews(tasks);
|
|
176
|
+
const reopen: string[] = [];
|
|
177
|
+
const reopenNode = (node: TaskView): void => {
|
|
178
|
+
if (node.status !== "pending") {
|
|
179
|
+
node.status = "pending";
|
|
180
|
+
node.skipReason = undefined;
|
|
181
|
+
reopen.push(node.id);
|
|
182
|
+
}
|
|
183
|
+
for (const child of node.children) reopenNode(child);
|
|
184
|
+
};
|
|
185
|
+
for (const node of flat) {
|
|
186
|
+
if (wanted.has(node.id)) reopenNode(node);
|
|
187
|
+
}
|
|
188
|
+
return reopen;
|
|
189
|
+
}
|
|
190
|
+
|
|
168
191
|
/** Checks that lose their satisfied state because a rollback reopened work
|
|
169
192
|
* they were verifying. Returns the ids whose `done` flag was cleared.
|
|
170
193
|
*
|
package/src/workflow-state.ts
CHANGED
|
@@ -156,8 +156,36 @@ export interface ExecutionCheckpoint {
|
|
|
156
156
|
* silently hand a stalled run a fresh budget; optional so checkpoints
|
|
157
157
|
* written before this field keep loading. */
|
|
158
158
|
stallRounds?: number;
|
|
159
|
+
/** v0.9.1 (F-002): set when the execution-review loop mechanically
|
|
160
|
+
* appended finding tasks to the approved plan. The checkpoint's plan
|
|
161
|
+
* identity is re-stamped at that moment so /resume-plans accepts the
|
|
162
|
+
* amended plan; the approval record keeps the ORIGINAL digest as
|
|
163
|
+
* evidence of what the user actually approved. */
|
|
164
|
+
planAmended?: { sha256: string; amendedAt: string; round: number };
|
|
159
165
|
/** v0.6.1: completion-audit bookkeeping (rounds, last failed set, pass). */
|
|
160
|
-
audit?: {
|
|
166
|
+
audit?: {
|
|
167
|
+
rounds: number;
|
|
168
|
+
lastResult?: string;
|
|
169
|
+
passed?: boolean;
|
|
170
|
+
undeterminable?: string[];
|
|
171
|
+
/** v0.9: unresolved implementation findings from the newest committed
|
|
172
|
+
* review round (stable F-### ids). Optional so pre-v0.9 checkpoints
|
|
173
|
+
* keep loading as "no findings". */
|
|
174
|
+
findings?: ReviewFindingRecord[];
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Serializable shape of one review finding (v0.9). Structurally identical to
|
|
179
|
+
* auditor.ts's ReviewFinding; declared here so the checkpoint layer does not
|
|
180
|
+
* import the auditor. */
|
|
181
|
+
export interface ReviewFindingRecord {
|
|
182
|
+
id: string;
|
|
183
|
+
severity: string;
|
|
184
|
+
taskIds: string[];
|
|
185
|
+
proposedTask?: string;
|
|
186
|
+
note: string;
|
|
187
|
+
evidence: string;
|
|
188
|
+
raw: string;
|
|
161
189
|
}
|
|
162
190
|
|
|
163
191
|
export interface OwnerInfo {
|
|
@@ -465,7 +493,7 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
465
493
|
const record = asRecord(value, label);
|
|
466
494
|
rejectExtraKeys(
|
|
467
495
|
record,
|
|
468
|
-
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "audit"]),
|
|
496
|
+
new Set(["approval", "doneVcIds", "implStatus", "currentI", "usage", "pausedReason", "reverifyAll", "originWorktree", "delegate", "tasks", "stallRounds", "planAmended", "audit"]),
|
|
469
497
|
label,
|
|
470
498
|
);
|
|
471
499
|
const execution: ExecutionCheckpoint = {
|
|
@@ -512,10 +540,19 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
512
540
|
if (record.stallRounds !== undefined && record.stallRounds !== null) {
|
|
513
541
|
execution.stallRounds = asInt(record.stallRounds, `${label}.stallRounds`, 0);
|
|
514
542
|
}
|
|
543
|
+
if (record.planAmended !== undefined && record.planAmended !== null) {
|
|
544
|
+
const rec = asRecord(record.planAmended, `${label}.planAmended`);
|
|
545
|
+
rejectExtraKeys(rec, new Set(["sha256", "amendedAt", "round"]), `${label}.planAmended`);
|
|
546
|
+
execution.planAmended = {
|
|
547
|
+
sha256: asString(rec.sha256, `${label}.planAmended.sha256`),
|
|
548
|
+
amendedAt: asString(rec.amendedAt, `${label}.planAmended.amendedAt`),
|
|
549
|
+
round: asInt(rec.round, `${label}.planAmended.round`, 0),
|
|
550
|
+
};
|
|
551
|
+
}
|
|
515
552
|
if (record.audit !== undefined && record.audit !== null) {
|
|
516
553
|
const audit = asRecord(record.audit, `${label}.audit`);
|
|
517
|
-
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable"]), `${label}.audit`);
|
|
518
|
-
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[] } = {
|
|
554
|
+
rejectExtraKeys(audit, new Set(["rounds", "lastResult", "passed", "undeterminable", "findings"]), `${label}.audit`);
|
|
555
|
+
const parsed: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] } = {
|
|
519
556
|
rounds: asInt(audit.rounds, `${label}.audit.rounds`, 0),
|
|
520
557
|
};
|
|
521
558
|
if (audit.lastResult !== undefined) parsed.lastResult = asString(audit.lastResult, `${label}.audit.lastResult`);
|
|
@@ -523,6 +560,24 @@ function asExecution(value: unknown, label: string): ExecutionCheckpoint {
|
|
|
523
560
|
if (audit.undeterminable !== undefined) {
|
|
524
561
|
parsed.undeterminable = asStringArray(audit.undeterminable, `${label}.audit.undeterminable`);
|
|
525
562
|
}
|
|
563
|
+
if (audit.findings !== undefined && audit.findings !== null) {
|
|
564
|
+
const arr = Array.isArray(audit.findings) ? audit.findings : null;
|
|
565
|
+
if (!arr) throw new CheckpointValidationError(`${label}.audit.findings must be an array`);
|
|
566
|
+
parsed.findings = arr.map((entry, i) => {
|
|
567
|
+
const rec = asRecord(entry, `${label}.audit.findings.${i}`);
|
|
568
|
+
rejectExtraKeys(rec, new Set(["id", "severity", "taskIds", "proposedTask", "note", "evidence", "raw"]), `${label}.audit.findings.${i}`);
|
|
569
|
+
const out: ReviewFindingRecord = {
|
|
570
|
+
id: asString(rec.id, `${label}.audit.findings.${i}.id`),
|
|
571
|
+
severity: asString(rec.severity, `${label}.audit.findings.${i}.severity`),
|
|
572
|
+
taskIds: rec.taskIds === undefined ? [] : asStringArray(rec.taskIds, `${label}.audit.findings.${i}.taskIds`),
|
|
573
|
+
note: rec.note === undefined ? "" : asString(rec.note, `${label}.audit.findings.${i}.note`),
|
|
574
|
+
evidence: rec.evidence === undefined ? "" : asString(rec.evidence, `${label}.audit.findings.${i}.evidence`),
|
|
575
|
+
raw: rec.raw === undefined ? "" : asString(rec.raw, `${label}.audit.findings.${i}.raw`),
|
|
576
|
+
};
|
|
577
|
+
if (rec.proposedTask !== undefined) out.proposedTask = asString(rec.proposedTask, `${label}.audit.findings.${i}.proposedTask`);
|
|
578
|
+
return out;
|
|
579
|
+
});
|
|
580
|
+
}
|
|
526
581
|
execution.audit = parsed;
|
|
527
582
|
}
|
|
528
583
|
return execution;
|
|
@@ -1080,8 +1135,10 @@ export interface ExecutionProgressInput {
|
|
|
1080
1135
|
delegate?: { modelSelector: string; startedAt: string } | null;
|
|
1081
1136
|
/** v0.6.1: task-tree progress snapshot (authoritative). */
|
|
1082
1137
|
tasks?: Record<string, { status: string; evidence?: string; skipReason?: string }>;
|
|
1083
|
-
/** v0.6.1: completion-audit bookkeeping update.
|
|
1084
|
-
|
|
1138
|
+
/** v0.6.1: completion-audit bookkeeping update. v0.9: findings rides the
|
|
1139
|
+
* same replace-semantics slot — callers that must preserve findings (e.g.
|
|
1140
|
+
* budget renewal) pass them through explicitly. */
|
|
1141
|
+
audit?: { rounds: number; lastResult?: string; passed?: boolean; undeterminable?: string[]; findings?: ReviewFindingRecord[] };
|
|
1085
1142
|
/** v0.7.1: watchdog budget counter, so a restart cannot refresh it. */
|
|
1086
1143
|
stallRounds?: number;
|
|
1087
1144
|
}
|
|
@@ -1108,6 +1165,23 @@ export function applyExecutionProgress(cp: WorkflowCheckpoint, progress: Executi
|
|
|
1108
1165
|
return { ...cp, execution };
|
|
1109
1166
|
}
|
|
1110
1167
|
|
|
1168
|
+
/** v0.9.1 (F-002): the execution-review loop appended finding tasks to the
|
|
1169
|
+
* approved plan file; re-stamp the checkpoint's plan identity to the amended
|
|
1170
|
+
* digest so a later /resume-plans does not reject the run as plan-mismatch.
|
|
1171
|
+
* The approval record is untouched — it keeps the digest the user actually
|
|
1172
|
+
* approved, and planAmended records when and why the identity moved. */
|
|
1173
|
+
export function applyExecutionPlanAmended(cp: WorkflowCheckpoint, plan: PlanIdentity, round: number): WorkflowCheckpoint {
|
|
1174
|
+
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
1175
|
+
return {
|
|
1176
|
+
...cp,
|
|
1177
|
+
plan,
|
|
1178
|
+
execution: {
|
|
1179
|
+
...cp.execution,
|
|
1180
|
+
planAmended: { sha256: plan.sha256, amendedAt: utcNow(), round },
|
|
1181
|
+
},
|
|
1182
|
+
};
|
|
1183
|
+
}
|
|
1184
|
+
|
|
1111
1185
|
/** D-011/F-001: code state changed under an unchanged plan — keep authorization, re-verify first. */
|
|
1112
1186
|
export function applyExecutionHeadChanged(cp: WorkflowCheckpoint): WorkflowCheckpoint {
|
|
1113
1187
|
if (cp.phase !== "executing" || !cp.execution) throw new StateError("requires phase \"executing\"");
|
|
@@ -303,7 +303,7 @@ describe("analyze_refs fanout", () => {
|
|
|
303
303
|
|
|
304
304
|
it("pins the per-batch overlay lifecycle (open before spawn, close in finally, cap 3)", () => {
|
|
305
305
|
const source = fs.readFileSync(path.join(ROOT, "tools", "analyze-refs.ts"), "utf8");
|
|
306
|
-
assert.equal((source.match(/new RefineOverlayController\("refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
|
|
306
|
+
assert.equal((source.match(/new RefineOverlayController\(\s*"refs"/g) ?? []).length, 1, "controller must be constructed per batch inside the loop");
|
|
307
307
|
assert.match(source, /overlay\?\.open\(refineOverlayContext\(ctx\), modelLabel\)/);
|
|
308
308
|
assert.match(source, /await overlay\?\.close\(\);/);
|
|
309
309
|
assert.match(source, /const BATCH_SIZE = 3;/);
|
package/tests/auditor.test.ts
CHANGED
|
@@ -2,6 +2,9 @@
|
|
|
2
2
|
* coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
|
|
3
3
|
|
|
4
4
|
import * as assert from "node:assert/strict";
|
|
5
|
+
import * as fs from "node:fs";
|
|
6
|
+
import * as os from "node:os";
|
|
7
|
+
import * as path from "node:path";
|
|
5
8
|
import { describe, it } from "node:test";
|
|
6
9
|
import type { CheckItem } from "../src/plan.ts";
|
|
7
10
|
import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
|
|
@@ -207,4 +210,185 @@ describe("audit rollback boundaries", () => {
|
|
|
207
210
|
assert.ok(presolved.includes("VC-003"));
|
|
208
211
|
assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
|
|
209
212
|
});
|
|
210
|
-
});
|
|
213
|
+
});
|
|
214
|
+
describe("findings parsing (v0.9)", () => {
|
|
215
|
+
const GOOD = [
|
|
216
|
+
"- `VC-001` — verdict: pass; evidence: ok",
|
|
217
|
+
"",
|
|
218
|
+
"- `F-001` — severity: high; tasks: Task-3, task-4; note: union rollback missing; evidence: src/exec.ts:1290",
|
|
219
|
+
"- `F-002` — severity: medium; tasks: none; proposed-task: cap retry backoff at 60s; note: unbounded; evidence: src/client.ts:12",
|
|
220
|
+
].join("\n");
|
|
221
|
+
|
|
222
|
+
it("parses well-formed findings with id/task normalization", () => {
|
|
223
|
+
const { findings } = parseAuditReport(GOOD, ALL_IDS);
|
|
224
|
+
assert.equal(findings.length, 2);
|
|
225
|
+
const f1 = findings[0];
|
|
226
|
+
assert.equal(f1.id, "F-001");
|
|
227
|
+
assert.equal(f1.severity, "high");
|
|
228
|
+
assert.deepEqual(f1.taskIds, ["Task-3", "Task-4"]);
|
|
229
|
+
assert.equal(f1.note, "union rollback missing");
|
|
230
|
+
assert.equal(f1.evidence, "src/exec.ts:1290");
|
|
231
|
+
const f2 = findings[1];
|
|
232
|
+
assert.equal(f2.severity, "medium");
|
|
233
|
+
assert.deepEqual(f2.taskIds, []);
|
|
234
|
+
assert.equal(f2.proposedTask, "cap retry backoff at 60s");
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
it("tolerates emphasis markers on severity", () => {
|
|
238
|
+
const { findings } = parseAuditReport("- `F-001` — severity: **high**; tasks: Task-1; note: n; evidence: e", ALL_IDS);
|
|
239
|
+
assert.equal(findings[0]?.severity, "high");
|
|
240
|
+
const { findings: f2 } = parseAuditReport("- `F-002` — severity: `medium`; tasks: none; note: n", ALL_IDS);
|
|
241
|
+
assert.equal(f2[0]?.severity, "medium");
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
it("degrades unreadable severity to a recorded non-blocking entry (never a rollback driver)", () => {
|
|
245
|
+
const { findings } = parseAuditReport("- `F-003` — tasks: whatever; note: no severity field", ALL_IDS);
|
|
246
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
247
|
+
assert.deepEqual(findings[0]?.taskIds, []);
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
it("degrades a missing mandatory tasks field", () => {
|
|
251
|
+
const { findings } = parseAuditReport("- `F-004` — severity: high; note: tasks field absent", ALL_IDS);
|
|
252
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
253
|
+
});
|
|
254
|
+
|
|
255
|
+
it("resolves duplicate ids to the first bullet", () => {
|
|
256
|
+
const report = [
|
|
257
|
+
"- `F-001` — severity: high; tasks: Task-1; note: first",
|
|
258
|
+
"- `F-001` — severity: low; tasks: none; note: second",
|
|
259
|
+
].join("\n");
|
|
260
|
+
const { findings } = parseAuditReport(report, ALL_IDS);
|
|
261
|
+
assert.equal(findings.length, 1);
|
|
262
|
+
assert.equal(findings[0].note, "first");
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
it("a finding bullet citing a VC verdict never registers that verdict", () => {
|
|
266
|
+
const report = "- `F-005` — severity: high; tasks: Task-1; note: cites VC-004 verdict: pass; evidence: z";
|
|
267
|
+
const { passed, failed } = parseAuditReport(report, ALL_IDS);
|
|
268
|
+
assert.equal(passed.length + failed.length, 0);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
it("prose mentioning F-### outside a bullet is ignored", () => {
|
|
272
|
+
const { findings } = parseAuditReport("also prose mentions F-009 not a bullet\n- `F-001` — severity: low; tasks: none; note: real", ALL_IDS);
|
|
273
|
+
assert.deepEqual(findings.map((f) => f.id), ["F-001"]);
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
it("applyAuditOutcome carries findings through to the outcome", () => {
|
|
277
|
+
const parsed = parseAuditReport(GOOD, ["VC-001"]);
|
|
278
|
+
const outcome = applyAuditOutcome(1, parsed, GOOD);
|
|
279
|
+
assert.equal(outcome.findings?.length, 2);
|
|
280
|
+
});
|
|
281
|
+
});
|
|
282
|
+
|
|
283
|
+
describe("section-aware verdict parsing (v0.9.1 F-011)", () => {
|
|
284
|
+
it("parses heading-style sections: id heading + verdict on its own line below", () => {
|
|
285
|
+
const report = [
|
|
286
|
+
"## 1. Verification verdicts",
|
|
287
|
+
"",
|
|
288
|
+
"### VC-001",
|
|
289
|
+
"- verdict: pass; evidence: src/a.ts",
|
|
290
|
+
"",
|
|
291
|
+
"### VC-002",
|
|
292
|
+
"- verdict: **fail**; evidence: src/c.ts",
|
|
293
|
+
"",
|
|
294
|
+
"### VC-003",
|
|
295
|
+
"verdict: undeterminable",
|
|
296
|
+
].join("\n");
|
|
297
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ["VC-001", "VC-002", "VC-003"]);
|
|
298
|
+
assert.deepEqual(passed, ["VC-001"]);
|
|
299
|
+
assert.deepEqual(failed, ["VC-002"]);
|
|
300
|
+
assert.deepEqual(undeterminable, ["VC-003"]);
|
|
301
|
+
});
|
|
302
|
+
|
|
303
|
+
it("a bare verdict with no open section is ignored (never misattributed)", () => {
|
|
304
|
+
const report = "Some preamble mentioning verdict: pass with no section above\n### VC-001\n- verdict: fail";
|
|
305
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
306
|
+
assert.deepEqual(passed, []);
|
|
307
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
308
|
+
});
|
|
309
|
+
|
|
310
|
+
it("an unknown section id does not capture later bare verdicts", () => {
|
|
311
|
+
const report = ["### VC-999", "- verdict: pass", "### VC-002", "- verdict: pass"].join("\n");
|
|
312
|
+
const { passed, undeterminable } = parseAuditReport(report, ["VC-002"]);
|
|
313
|
+
assert.deepEqual(passed, ["VC-002"]);
|
|
314
|
+
assert.deepEqual(undeterminable, []);
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
it("finding bullets never open a verdict section", () => {
|
|
318
|
+
const report = [
|
|
319
|
+
"### VC-001",
|
|
320
|
+
"- `F-005` — severity: high; tasks: Task-1; note: cites verdict: pass inside a note; evidence: e",
|
|
321
|
+
"- verdict: fail",
|
|
322
|
+
].join("\n");
|
|
323
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
324
|
+
assert.deepEqual(passed, []);
|
|
325
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
326
|
+
assert.equal(parseAuditReport(report, ["VC-001"]).findings[0]?.severity, "high");
|
|
327
|
+
});
|
|
328
|
+
|
|
329
|
+
it("knownTaskIds filters finding mappings to plan tasks (F-003)", () => {
|
|
330
|
+
const report = "- `F-001` — severity: high; tasks: Task-1, VC-007, bogus; note: n; evidence: e";
|
|
331
|
+
const { findings } = parseAuditReport(report, [], new Set(["Task-1"]));
|
|
332
|
+
assert.deepEqual(findings[0]?.taskIds, ["Task-1"]);
|
|
333
|
+
});
|
|
334
|
+
});
|
|
335
|
+
|
|
336
|
+
describe("review brief dual-output contract (v0.9)", () => {
|
|
337
|
+
it("demands both sections: verdicts and findings grammar", () => {
|
|
338
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
339
|
+
assert.match(task, /1\. Verification verdicts/);
|
|
340
|
+
assert.match(task, /2\. Implementation findings/);
|
|
341
|
+
assert.match(task, /severity: high \| medium \| low/);
|
|
342
|
+
assert.match(task, /proposed-task:/);
|
|
343
|
+
});
|
|
344
|
+
|
|
345
|
+
it("lists plan tasks as the valid mapping domain", () => {
|
|
346
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
347
|
+
assert.match(task, /Plan tasks \(the only ids valid in a finding's tasks field\):/);
|
|
348
|
+
assert.match(task, /`Task-3\.1`: injection/);
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
it("injects prior unresolved findings with the stable-id reuse instruction", () => {
|
|
352
|
+
const prior = [{
|
|
353
|
+
id: "F-001", severity: "high" as const, taskIds: ["Task-2"], note: "still broken",
|
|
354
|
+
evidence: "src/b.ts", raw: "- `F-001` — severity: high; tasks: Task-2; note: still broken",
|
|
355
|
+
}];
|
|
356
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 3, prior);
|
|
357
|
+
assert.match(task, /reuse these exact ids while the problem persists/);
|
|
358
|
+
assert.match(task, /`F-001` — severity: high; tasks: Task-2/);
|
|
359
|
+
});
|
|
360
|
+
|
|
361
|
+
it("marks the first findings round when no prior list exists", () => {
|
|
362
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
363
|
+
assert.match(task, /first round with findings in scope/);
|
|
364
|
+
});
|
|
365
|
+
});
|
|
366
|
+
|
|
367
|
+
describe("review round report findings lines (v0.9)", () => {
|
|
368
|
+
it("emits the high-findings line and per-finding detail", async () => {
|
|
369
|
+
const { writeReviewRoundReport } = await import("../src/auditor.ts");
|
|
370
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-round-report-"));
|
|
371
|
+
try {
|
|
372
|
+
const file = writeReviewRoundReport(dir, {
|
|
373
|
+
budgetRound: 2,
|
|
374
|
+
attempt: 1,
|
|
375
|
+
outcome: "failed",
|
|
376
|
+
passed: ["VC-001"],
|
|
377
|
+
failed: [],
|
|
378
|
+
undeterminable: [],
|
|
379
|
+
findings: [
|
|
380
|
+
{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "e", raw: "raw" },
|
|
381
|
+
{ id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy up", note: "polish", evidence: "e", raw: "raw" },
|
|
382
|
+
],
|
|
383
|
+
coveredTaskIds: ["Task-1", "Task-2"],
|
|
384
|
+
report: "## Report\nbody",
|
|
385
|
+
});
|
|
386
|
+
assert.ok(file);
|
|
387
|
+
const text = fs.readFileSync(file, "utf8");
|
|
388
|
+
assert.match(text, /- high findings: F-001/);
|
|
389
|
+
assert.match(text, /F-002 \(medium; proposed: tidy up\)/);
|
|
390
|
+
} finally {
|
|
391
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
392
|
+
}
|
|
393
|
+
});
|
|
394
|
+
});
|
package/tests/dashboard.test.ts
CHANGED
|
@@ -14,7 +14,7 @@ import * as assert from "node:assert/strict";
|
|
|
14
14
|
import { describe, it } from "node:test";
|
|
15
15
|
import { visibleWidth } from "@earendil-works/pi-tui";
|
|
16
16
|
import { parsePlanTasks } from "../src/plan.ts";
|
|
17
|
-
import { buildTaskView } from "../src/tasks.ts";
|
|
17
|
+
import { buildTaskView, flattenTaskViews } from "../src/tasks.ts";
|
|
18
18
|
import { visibleWidth as localVisibleWidth } from "../src/refine-ui-helpers.ts";
|
|
19
19
|
import {
|
|
20
20
|
deriveDashboardModel,
|
|
@@ -400,3 +400,69 @@ describe("rolled-back task rendering", () => {
|
|
|
400
400
|
}
|
|
401
401
|
});
|
|
402
402
|
});
|
|
403
|
+
|
|
404
|
+
describe("findings visibility (v0.9)", () => {
|
|
405
|
+
const findings = [
|
|
406
|
+
{ id: "F-001", severity: "high", note: "union rollback missing", taskIds: ["Task-3"] },
|
|
407
|
+
{ id: "F-002", severity: "medium", note: "polish", taskIds: [] },
|
|
408
|
+
];
|
|
409
|
+
|
|
410
|
+
function withFindings(extra: { auditRounds?: number; reviewRunning?: boolean }) {
|
|
411
|
+
const tasks = buildTaskView(parsePlanTasks(PLAN), {});
|
|
412
|
+
const checklist = [
|
|
413
|
+
{ id: "VC-001", text: "`VC-001` covers `Task-1`; pass condition: x", done: false },
|
|
414
|
+
{ id: "VC-002", text: "`VC-002` covers `Task-3`; pass condition: y", done: false },
|
|
415
|
+
];
|
|
416
|
+
return deriveDashboardModel("demo-run", tasks, checklist, {
|
|
417
|
+
startedAt: new Date().toISOString(),
|
|
418
|
+
findings,
|
|
419
|
+
auditRounds: extra.auditRounds ?? null,
|
|
420
|
+
reviewRunning: extra.reviewRunning ?? false,
|
|
421
|
+
});
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
it("summary line shows review round/5 and the high count", () => {
|
|
425
|
+
const line = formatDashboardSummaryLine(withFindings({ auditRounds: 2 }));
|
|
426
|
+
assert.match(line, /review r2\/5/);
|
|
427
|
+
assert.match(line, /1 high/);
|
|
428
|
+
});
|
|
429
|
+
|
|
430
|
+
it("summary line omits the high token when only non-high findings remain", () => {
|
|
431
|
+
const tasks = buildTaskView(parsePlanTasks(PLAN), {});
|
|
432
|
+
const m = deriveDashboardModel("demo-run", tasks, [], {
|
|
433
|
+
startedAt: new Date().toISOString(),
|
|
434
|
+
findings: [{ id: "F-002", severity: "medium", note: "polish", taskIds: [] }],
|
|
435
|
+
auditRounds: 3,
|
|
436
|
+
});
|
|
437
|
+
const line = formatDashboardSummaryLine(m);
|
|
438
|
+
assert.match(line, /review r3\/5/);
|
|
439
|
+
assert.doesNotMatch(line, /high/);
|
|
440
|
+
});
|
|
441
|
+
|
|
442
|
+
it("compact panel renders the high-findings line in BOTH phases", () => {
|
|
443
|
+
// Non-terminal phase (the executor is repairing): the line must show.
|
|
444
|
+
const repairing = renderDashboardLines(withFindings({ auditRounds: 1 }), 80);
|
|
445
|
+
assert.ok(repairing.some((l) => /⚠.*high: F-001/.test(l)), "high findings visible while repairing");
|
|
446
|
+
|
|
447
|
+
// Terminal phase: still visible alongside the round counter.
|
|
448
|
+
const everyId = Object.fromEntries(
|
|
449
|
+
flattenTaskViews(buildTaskView(parsePlanTasks(PLAN), {})).map((t) => [t.id, { status: "complete" as const }]),
|
|
450
|
+
);
|
|
451
|
+
const allDone = buildTaskView(parsePlanTasks(PLAN), everyId);
|
|
452
|
+
const terminal = deriveDashboardModel("demo-run", allDone, [], {
|
|
453
|
+
startedAt: new Date().toISOString(),
|
|
454
|
+
findings,
|
|
455
|
+
auditRounds: 1,
|
|
456
|
+
});
|
|
457
|
+
const lines = renderDashboardLines(terminal, 80);
|
|
458
|
+
assert.ok(lines.some((l) => /⚠.*high: F-001/.test(l)));
|
|
459
|
+
assert.ok(lines.some((l) => /high finding\(s\) unresolved/.test(l)));
|
|
460
|
+
});
|
|
461
|
+
|
|
462
|
+
it("tree view lists findings with severity and mapping, and the verdict line names unresolved highs", () => {
|
|
463
|
+
const lines = renderDashboardTreeLines(withFindings({ auditRounds: 2 }), 100);
|
|
464
|
+
assert.ok(lines.some((l) => /⚠ F-001 \(high, Task-3\): union rollback missing/.test(l)));
|
|
465
|
+
assert.ok(lines.some((l) => /· F-002 \(medium\): polish/.test(l)));
|
|
466
|
+
assert.ok(lines.some((l) => /high findings unresolved: F-001/.test(l)));
|
|
467
|
+
});
|
|
468
|
+
});
|