pi-plans 0.7.0 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +92 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +14 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +306 -63
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +91 -13
- package/src/exec.ts +835 -142
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +40 -8
- package/src/resume-command.ts +19 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +62 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +93 -6
- package/tests/analyze-refs.test.ts +1 -1
- package/tests/auditor.test.ts +299 -16
- package/tests/dashboard.test.ts +202 -2
- package/tests/exec-review-loop.test.ts +724 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/refine-ui.test.ts +25 -2
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +105 -0
- package/tools/analyze-refs.ts +17 -6
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
- package/tools/refine.ts +22 -3
package/tests/auditor.test.ts
CHANGED
|
@@ -1,11 +1,20 @@
|
|
|
1
|
-
/** Tests for the
|
|
2
|
-
* boundaries, skipped-pass, and no-cover exclusion. */
|
|
1
|
+
/** Tests for the execution reviewer: tri-state verdict parsing, contract
|
|
2
|
+
* coverage, rollback boundaries, skipped-pass, and no-cover exclusion. */
|
|
3
3
|
|
|
4
4
|
import * as assert from "node:assert/strict";
|
|
5
|
+
import * as fs from "node:fs";
|
|
6
|
+
import * as os from "node:os";
|
|
7
|
+
import * as path from "node:path";
|
|
5
8
|
import { describe, it } from "node:test";
|
|
6
9
|
import type { CheckItem } from "../src/plan.ts";
|
|
7
|
-
import { buildTaskView } from "../src/tasks.ts";
|
|
8
|
-
import {
|
|
10
|
+
import { auditRollbackSet, buildTaskView } from "../src/tasks.ts";
|
|
11
|
+
import {
|
|
12
|
+
applyAuditOutcome,
|
|
13
|
+
auditablePendingChecks,
|
|
14
|
+
buildAuditTask,
|
|
15
|
+
parseAuditReport,
|
|
16
|
+
presolvedCheckIds,
|
|
17
|
+
} from "../src/auditor.ts";
|
|
9
18
|
import { parsePlanTasks } from "../src/plan.ts";
|
|
10
19
|
|
|
11
20
|
const PLAN = `## Tasks
|
|
@@ -37,6 +46,8 @@ function view(progress?: Record<string, { status: "complete" | "skipped" }>) {
|
|
|
37
46
|
return buildTaskView(parsePlanTasks(PLAN), progress);
|
|
38
47
|
}
|
|
39
48
|
|
|
49
|
+
const ALL_IDS = ["VC-001", "VC-002", "VC-003", "VC-004"];
|
|
50
|
+
|
|
40
51
|
describe("auditor parsing", () => {
|
|
41
52
|
it("parses per-check verdicts and ignores unknown ids", () => {
|
|
42
53
|
const report = [
|
|
@@ -44,48 +55,139 @@ describe("auditor parsing", () => {
|
|
|
44
55
|
"- `VC-002` — verdict: fail; evidence: src/c.ts; note: missing",
|
|
45
56
|
"- `VC-999` — verdict: pass; evidence: n/a",
|
|
46
57
|
].join("\n");
|
|
47
|
-
const { passed, failed } = parseAuditReport(report,
|
|
58
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
48
59
|
assert.deepEqual(passed, ["VC-001"]);
|
|
49
60
|
assert.deepEqual(failed, ["VC-002"]);
|
|
61
|
+
assert.deepEqual(undeterminable.sort(), ["VC-003", "VC-004"]);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
it("parses verdicts wrapped in markdown emphasis", () => {
|
|
65
|
+
// Regression: `verdict: **pass**` used to parse as nothing, and the
|
|
66
|
+
// caller's fail-closed rule then rolled the entire run back.
|
|
67
|
+
const report = [
|
|
68
|
+
"- `VC-001` — verdict: **pass**; evidence: src/a.ts",
|
|
69
|
+
"- `VC-002` — verdict: **fail**; evidence: src/c.ts",
|
|
70
|
+
"- `VC-003` — verdict: *pass*; evidence: src/d.ts",
|
|
71
|
+
"- `VC-004` — verdict: `pass`; evidence: src/e.ts",
|
|
72
|
+
].join("\n");
|
|
73
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
74
|
+
assert.deepEqual(passed, ["VC-001", "VC-003", "VC-004"]);
|
|
75
|
+
assert.deepEqual(failed, ["VC-002"]);
|
|
76
|
+
assert.deepEqual(undeterminable, []);
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
it("does not read a verdict word prefix as a verdict", () => {
|
|
80
|
+
const report = ["- `VC-001` — verdict: passed; evidence: x", "- `VC-002` — verdict: failing; evidence: y"].join("\n");
|
|
81
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
82
|
+
assert.deepEqual(passed, []);
|
|
83
|
+
assert.deepEqual(failed, []);
|
|
84
|
+
assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
|
|
50
85
|
});
|
|
51
86
|
|
|
52
87
|
it("conflicting verdicts for one check resolve to fail", () => {
|
|
53
88
|
const report = "- `VC-001` — verdict: pass; note: a\n- `VC-001` — verdict: fail; note: b";
|
|
54
|
-
const { passed, failed } = parseAuditReport(report,
|
|
89
|
+
const { passed, failed } = parseAuditReport(report, ALL_IDS);
|
|
55
90
|
assert.deepEqual(passed, []);
|
|
56
91
|
assert.deepEqual(failed, ["VC-001"]);
|
|
57
92
|
});
|
|
58
93
|
|
|
59
|
-
it("
|
|
94
|
+
it("reads an explicit undeterminable verdict", () => {
|
|
95
|
+
const report = "- `VC-001` — verdict: pass\n- `VC-002` — verdict: undeterminable; evidence: bash not granted";
|
|
96
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
97
|
+
assert.deepEqual(passed, ["VC-001"]);
|
|
98
|
+
assert.deepEqual(failed, []);
|
|
99
|
+
assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-003", "VC-004"]);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
it("treats a reviewer-shaped report as undeterminable, never as failure", () => {
|
|
103
|
+
// Regression: agents/reviewer.md was the auditor's system prompt and
|
|
104
|
+
// mandates `## Findings` / `## Questions` with F-### lines and never
|
|
105
|
+
// says "verdict". Under fail-closed that whole shape read as "every
|
|
106
|
+
// check failed" and rolled the run back.
|
|
107
|
+
const report = [
|
|
108
|
+
"## Findings",
|
|
109
|
+
"",
|
|
110
|
+
"- `F-001` — severity: high; affected: VC-001; evidence: src/a.ts; impact: x; recommended fix: y; suggested disposition: accept.",
|
|
111
|
+
"",
|
|
112
|
+
"## Questions",
|
|
113
|
+
"",
|
|
114
|
+
"None.",
|
|
115
|
+
].join("\n");
|
|
116
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
117
|
+
assert.deepEqual(passed, []);
|
|
118
|
+
assert.deepEqual(failed, [], "a reviewer-shaped report is not a failed audit");
|
|
119
|
+
assert.deepEqual(undeterminable.sort(), ["VC-001", "VC-002", "VC-003", "VC-004"]);
|
|
120
|
+
});
|
|
121
|
+
|
|
122
|
+
it("classifies a partially covered report per check", () => {
|
|
123
|
+
const report = "- `VC-001` — verdict: pass\n- `VC-003` — verdict: **pass**";
|
|
124
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ALL_IDS);
|
|
125
|
+
assert.deepEqual(passed, ["VC-001", "VC-003"]);
|
|
126
|
+
assert.deepEqual(failed, []);
|
|
127
|
+
assert.deepEqual(undeterminable.sort(), ["VC-002", "VC-004"]);
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
it("builds the brief with the auditable pending checks only", () => {
|
|
60
131
|
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
61
132
|
assert.match(task, /read-only/i);
|
|
62
133
|
assert.match(task, /`VC-001`/);
|
|
63
134
|
assert.doesNotMatch(task, /`VC-999`/);
|
|
135
|
+
assert.doesNotMatch(task, /`VC-004`/, "a check covering no task is never audited");
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
it("brief states the tri-state contract and forbids fail-for-want-of-evidence", () => {
|
|
139
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
140
|
+
assert.match(task, /verdict: pass \| fail \| undeterminable/);
|
|
141
|
+
assert.match(task, /never report `fail` for want of evidence/i);
|
|
142
|
+
});
|
|
143
|
+
|
|
144
|
+
it("brief omits checks already done in an earlier round", () => {
|
|
145
|
+
// Regression: the brief listed every auditable check, so a round-2
|
|
146
|
+
// auditor that legitimately skipped the satisfied ones looked like a
|
|
147
|
+
// contract violation -- and, with no rollback to produce progress, the
|
|
148
|
+
// budget climbed to the cap and the run livelocked.
|
|
149
|
+
const done = buildAuditTask("/tmp/PLAN_v1.md", checks(["VC-001"]), view(), 2);
|
|
150
|
+
assert.doesNotMatch(done, /`VC-001`/);
|
|
151
|
+
assert.match(done, /`VC-002`/);
|
|
152
|
+
assert.deepEqual(
|
|
153
|
+
auditablePendingChecks(checks(["VC-001"]), view()).map((item) => item.id),
|
|
154
|
+
["VC-002", "VC-003"],
|
|
155
|
+
);
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
it("applyAuditOutcome classifies without touching the checklist or task tree", () => {
|
|
159
|
+
const checklist = checks();
|
|
160
|
+
const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
161
|
+
const parsed = parseAuditReport("- `VC-001` — verdict: pass\n- `VC-002` — verdict: fail", ["VC-001", "VC-002"]);
|
|
162
|
+
const outcome = applyAuditOutcome(3, parsed, "report text");
|
|
163
|
+
assert.deepEqual(outcome.passed, ["VC-001"]);
|
|
164
|
+
assert.deepEqual(outcome.failed, ["VC-002"]);
|
|
165
|
+
assert.deepEqual(outcome.undeterminable, []);
|
|
166
|
+
assert.equal(outcome.round, 3);
|
|
167
|
+
assert.equal(outcome.report, "report text");
|
|
168
|
+
assert.equal(checklist.every((item) => item.done === false), true, "no done flag written here");
|
|
169
|
+
assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "complete", "no rollback here");
|
|
64
170
|
});
|
|
65
171
|
});
|
|
66
172
|
|
|
67
173
|
describe("audit rollback boundaries", () => {
|
|
68
174
|
it("failed multi-cover check rolls back every covered task", () => {
|
|
69
175
|
const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
70
|
-
const
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
// Passed check stays done; its task stays closed.
|
|
74
|
-
assert.equal(checks().length, 4);
|
|
176
|
+
const rolledBack = auditRollbackSet(tasks, checks(), "VC-002");
|
|
177
|
+
assert.deepEqual(rolledBack.sort(), ["Task-2", "Task-3", "Task-3.1"]);
|
|
178
|
+
// A check that passed keeps its task closed.
|
|
75
179
|
assert.equal(tasks.find((t) => t.id === "Task-1")?.status, "complete");
|
|
76
180
|
assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "pending");
|
|
77
181
|
});
|
|
78
182
|
|
|
79
183
|
it("covering the parent cascades to subtasks even when the parent alone is listed", () => {
|
|
80
184
|
const tasks = view({ "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
|
|
81
|
-
|
|
82
|
-
assert.ok(outcome.rolledBack.includes("Task-3.1"), "subtask reopens with the parent");
|
|
185
|
+
assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-3.1"), "subtask reopens with the parent");
|
|
83
186
|
});
|
|
84
187
|
|
|
85
188
|
it("skipped tasks reopen too (skip state cleared)", () => {
|
|
86
189
|
const tasks = view({ "Task-2": { status: "skipped" } });
|
|
87
|
-
|
|
88
|
-
assert.ok(outcome.rolledBack.includes("Task-2"));
|
|
190
|
+
assert.ok(auditRollbackSet(tasks, checks(), "VC-002").includes("Task-2"));
|
|
89
191
|
assert.equal(tasks.find((t) => t.id === "Task-2")?.skipReason, undefined);
|
|
90
192
|
});
|
|
91
193
|
|
|
@@ -109,3 +211,184 @@ describe("audit rollback boundaries", () => {
|
|
|
109
211
|
assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
|
|
110
212
|
});
|
|
111
213
|
});
|
|
214
|
+
describe("findings parsing (v0.9)", () => {
|
|
215
|
+
const GOOD = [
|
|
216
|
+
"- `VC-001` — verdict: pass; evidence: ok",
|
|
217
|
+
"",
|
|
218
|
+
"- `F-001` — severity: high; tasks: Task-3, task-4; note: union rollback missing; evidence: src/exec.ts:1290",
|
|
219
|
+
"- `F-002` — severity: medium; tasks: none; proposed-task: cap retry backoff at 60s; note: unbounded; evidence: src/client.ts:12",
|
|
220
|
+
].join("\n");
|
|
221
|
+
|
|
222
|
+
it("parses well-formed findings with id/task normalization", () => {
|
|
223
|
+
const { findings } = parseAuditReport(GOOD, ALL_IDS);
|
|
224
|
+
assert.equal(findings.length, 2);
|
|
225
|
+
const f1 = findings[0];
|
|
226
|
+
assert.equal(f1.id, "F-001");
|
|
227
|
+
assert.equal(f1.severity, "high");
|
|
228
|
+
assert.deepEqual(f1.taskIds, ["Task-3", "Task-4"]);
|
|
229
|
+
assert.equal(f1.note, "union rollback missing");
|
|
230
|
+
assert.equal(f1.evidence, "src/exec.ts:1290");
|
|
231
|
+
const f2 = findings[1];
|
|
232
|
+
assert.equal(f2.severity, "medium");
|
|
233
|
+
assert.deepEqual(f2.taskIds, []);
|
|
234
|
+
assert.equal(f2.proposedTask, "cap retry backoff at 60s");
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
it("tolerates emphasis markers on severity", () => {
|
|
238
|
+
const { findings } = parseAuditReport("- `F-001` — severity: **high**; tasks: Task-1; note: n; evidence: e", ALL_IDS);
|
|
239
|
+
assert.equal(findings[0]?.severity, "high");
|
|
240
|
+
const { findings: f2 } = parseAuditReport("- `F-002` — severity: `medium`; tasks: none; note: n", ALL_IDS);
|
|
241
|
+
assert.equal(f2[0]?.severity, "medium");
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
it("degrades unreadable severity to a recorded non-blocking entry (never a rollback driver)", () => {
|
|
245
|
+
const { findings } = parseAuditReport("- `F-003` — tasks: whatever; note: no severity field", ALL_IDS);
|
|
246
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
247
|
+
assert.deepEqual(findings[0]?.taskIds, []);
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
it("degrades a missing mandatory tasks field", () => {
|
|
251
|
+
const { findings } = parseAuditReport("- `F-004` — severity: high; note: tasks field absent", ALL_IDS);
|
|
252
|
+
assert.equal(findings[0]?.severity, "malformed");
|
|
253
|
+
});
|
|
254
|
+
|
|
255
|
+
it("resolves duplicate ids to the first bullet", () => {
|
|
256
|
+
const report = [
|
|
257
|
+
"- `F-001` — severity: high; tasks: Task-1; note: first",
|
|
258
|
+
"- `F-001` — severity: low; tasks: none; note: second",
|
|
259
|
+
].join("\n");
|
|
260
|
+
const { findings } = parseAuditReport(report, ALL_IDS);
|
|
261
|
+
assert.equal(findings.length, 1);
|
|
262
|
+
assert.equal(findings[0].note, "first");
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
it("a finding bullet citing a VC verdict never registers that verdict", () => {
|
|
266
|
+
const report = "- `F-005` — severity: high; tasks: Task-1; note: cites VC-004 verdict: pass; evidence: z";
|
|
267
|
+
const { passed, failed } = parseAuditReport(report, ALL_IDS);
|
|
268
|
+
assert.equal(passed.length + failed.length, 0);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
it("prose mentioning F-### outside a bullet is ignored", () => {
|
|
272
|
+
const { findings } = parseAuditReport("also prose mentions F-009 not a bullet\n- `F-001` — severity: low; tasks: none; note: real", ALL_IDS);
|
|
273
|
+
assert.deepEqual(findings.map((f) => f.id), ["F-001"]);
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
it("applyAuditOutcome carries findings through to the outcome", () => {
|
|
277
|
+
const parsed = parseAuditReport(GOOD, ["VC-001"]);
|
|
278
|
+
const outcome = applyAuditOutcome(1, parsed, GOOD);
|
|
279
|
+
assert.equal(outcome.findings?.length, 2);
|
|
280
|
+
});
|
|
281
|
+
});
|
|
282
|
+
|
|
283
|
+
describe("section-aware verdict parsing (v0.9.1 F-011)", () => {
|
|
284
|
+
it("parses heading-style sections: id heading + verdict on its own line below", () => {
|
|
285
|
+
const report = [
|
|
286
|
+
"## 1. Verification verdicts",
|
|
287
|
+
"",
|
|
288
|
+
"### VC-001",
|
|
289
|
+
"- verdict: pass; evidence: src/a.ts",
|
|
290
|
+
"",
|
|
291
|
+
"### VC-002",
|
|
292
|
+
"- verdict: **fail**; evidence: src/c.ts",
|
|
293
|
+
"",
|
|
294
|
+
"### VC-003",
|
|
295
|
+
"verdict: undeterminable",
|
|
296
|
+
].join("\n");
|
|
297
|
+
const { passed, failed, undeterminable } = parseAuditReport(report, ["VC-001", "VC-002", "VC-003"]);
|
|
298
|
+
assert.deepEqual(passed, ["VC-001"]);
|
|
299
|
+
assert.deepEqual(failed, ["VC-002"]);
|
|
300
|
+
assert.deepEqual(undeterminable, ["VC-003"]);
|
|
301
|
+
});
|
|
302
|
+
|
|
303
|
+
it("a bare verdict with no open section is ignored (never misattributed)", () => {
|
|
304
|
+
const report = "Some preamble mentioning verdict: pass with no section above\n### VC-001\n- verdict: fail";
|
|
305
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
306
|
+
assert.deepEqual(passed, []);
|
|
307
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
308
|
+
});
|
|
309
|
+
|
|
310
|
+
it("an unknown section id does not capture later bare verdicts", () => {
|
|
311
|
+
const report = ["### VC-999", "- verdict: pass", "### VC-002", "- verdict: pass"].join("\n");
|
|
312
|
+
const { passed, undeterminable } = parseAuditReport(report, ["VC-002"]);
|
|
313
|
+
assert.deepEqual(passed, ["VC-002"]);
|
|
314
|
+
assert.deepEqual(undeterminable, []);
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
it("finding bullets never open a verdict section", () => {
|
|
318
|
+
const report = [
|
|
319
|
+
"### VC-001",
|
|
320
|
+
"- `F-005` — severity: high; tasks: Task-1; note: cites verdict: pass inside a note; evidence: e",
|
|
321
|
+
"- verdict: fail",
|
|
322
|
+
].join("\n");
|
|
323
|
+
const { passed, failed } = parseAuditReport(report, ["VC-001"]);
|
|
324
|
+
assert.deepEqual(passed, []);
|
|
325
|
+
assert.deepEqual(failed, ["VC-001"]);
|
|
326
|
+
assert.equal(parseAuditReport(report, ["VC-001"]).findings[0]?.severity, "high");
|
|
327
|
+
});
|
|
328
|
+
|
|
329
|
+
it("knownTaskIds filters finding mappings to plan tasks (F-003)", () => {
|
|
330
|
+
const report = "- `F-001` — severity: high; tasks: Task-1, VC-007, bogus; note: n; evidence: e";
|
|
331
|
+
const { findings } = parseAuditReport(report, [], new Set(["Task-1"]));
|
|
332
|
+
assert.deepEqual(findings[0]?.taskIds, ["Task-1"]);
|
|
333
|
+
});
|
|
334
|
+
});
|
|
335
|
+
|
|
336
|
+
describe("review brief dual-output contract (v0.9)", () => {
|
|
337
|
+
it("demands both sections: verdicts and findings grammar", () => {
|
|
338
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
339
|
+
assert.match(task, /1\. Verification verdicts/);
|
|
340
|
+
assert.match(task, /2\. Implementation findings/);
|
|
341
|
+
assert.match(task, /severity: high \| medium \| low/);
|
|
342
|
+
assert.match(task, /proposed-task:/);
|
|
343
|
+
});
|
|
344
|
+
|
|
345
|
+
it("lists plan tasks as the valid mapping domain", () => {
|
|
346
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
347
|
+
assert.match(task, /Plan tasks \(the only ids valid in a finding's tasks field\):/);
|
|
348
|
+
assert.match(task, /`Task-3\.1`: injection/);
|
|
349
|
+
});
|
|
350
|
+
|
|
351
|
+
it("injects prior unresolved findings with the stable-id reuse instruction", () => {
|
|
352
|
+
const prior = [{
|
|
353
|
+
id: "F-001", severity: "high" as const, taskIds: ["Task-2"], note: "still broken",
|
|
354
|
+
evidence: "src/b.ts", raw: "- `F-001` — severity: high; tasks: Task-2; note: still broken",
|
|
355
|
+
}];
|
|
356
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 3, prior);
|
|
357
|
+
assert.match(task, /reuse these exact ids while the problem persists/);
|
|
358
|
+
assert.match(task, /`F-001` — severity: high; tasks: Task-2/);
|
|
359
|
+
});
|
|
360
|
+
|
|
361
|
+
it("marks the first findings round when no prior list exists", () => {
|
|
362
|
+
const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
|
|
363
|
+
assert.match(task, /first round with findings in scope/);
|
|
364
|
+
});
|
|
365
|
+
});
|
|
366
|
+
|
|
367
|
+
describe("review round report findings lines (v0.9)", () => {
|
|
368
|
+
it("emits the high-findings line and per-finding detail", async () => {
|
|
369
|
+
const { writeReviewRoundReport } = await import("../src/auditor.ts");
|
|
370
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-round-report-"));
|
|
371
|
+
try {
|
|
372
|
+
const file = writeReviewRoundReport(dir, {
|
|
373
|
+
budgetRound: 2,
|
|
374
|
+
attempt: 1,
|
|
375
|
+
outcome: "failed",
|
|
376
|
+
passed: ["VC-001"],
|
|
377
|
+
failed: [],
|
|
378
|
+
undeterminable: [],
|
|
379
|
+
findings: [
|
|
380
|
+
{ id: "F-001", severity: "high", taskIds: ["Task-2"], note: "broken", evidence: "e", raw: "raw" },
|
|
381
|
+
{ id: "F-002", severity: "medium", taskIds: [], proposedTask: "tidy up", note: "polish", evidence: "e", raw: "raw" },
|
|
382
|
+
],
|
|
383
|
+
coveredTaskIds: ["Task-1", "Task-2"],
|
|
384
|
+
report: "## Report\nbody",
|
|
385
|
+
});
|
|
386
|
+
assert.ok(file);
|
|
387
|
+
const text = fs.readFileSync(file, "utf8");
|
|
388
|
+
assert.match(text, /- high findings: F-001/);
|
|
389
|
+
assert.match(text, /F-002 \(medium; proposed: tidy up\)/);
|
|
390
|
+
} finally {
|
|
391
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
392
|
+
}
|
|
393
|
+
});
|
|
394
|
+
});
|
package/tests/dashboard.test.ts
CHANGED
|
@@ -14,7 +14,7 @@ import * as assert from "node:assert/strict";
|
|
|
14
14
|
import { describe, it } from "node:test";
|
|
15
15
|
import { visibleWidth } from "@earendil-works/pi-tui";
|
|
16
16
|
import { parsePlanTasks } from "../src/plan.ts";
|
|
17
|
-
import { buildTaskView } from "../src/tasks.ts";
|
|
17
|
+
import { buildTaskView, flattenTaskViews } from "../src/tasks.ts";
|
|
18
18
|
import { visibleWidth as localVisibleWidth } from "../src/refine-ui-helpers.ts";
|
|
19
19
|
import {
|
|
20
20
|
deriveDashboardModel,
|
|
@@ -134,7 +134,7 @@ describe("expanded tree rendering", () => {
|
|
|
134
134
|
m.auditRounds = 2;
|
|
135
135
|
m.auditFailed = ["VC-001"];
|
|
136
136
|
const lines = renderDashboardTreeLines(m, 100);
|
|
137
|
-
assert.ok(lines.some((line) => line.includes("
|
|
137
|
+
assert.ok(lines.some((line) => line.includes("Execution review: round 2")));
|
|
138
138
|
});
|
|
139
139
|
});
|
|
140
140
|
|
|
@@ -266,3 +266,203 @@ describe("width invariant (TUI crash regression)", () => {
|
|
|
266
266
|
}
|
|
267
267
|
});
|
|
268
268
|
});
|
|
269
|
+
|
|
270
|
+
describe("rolled-back task rendering", () => {
|
|
271
|
+
const CHECKS = () => [
|
|
272
|
+
{ id: "VC-001", text: "`VC-001` covers `Task-1`; pass condition: x", done: false },
|
|
273
|
+
{ id: "VC-002", text: "`VC-002` covers `Task-3`; pass condition: y", done: false },
|
|
274
|
+
];
|
|
275
|
+
|
|
276
|
+
function withProgress(progress: Record<string, { status: "complete" | "skipped" | "pending"; evidence?: string }>) {
|
|
277
|
+
return deriveDashboardModel("demo-run", buildTaskView(parsePlanTasks(PLAN), progress), CHECKS(), {
|
|
278
|
+
startedAt: new Date().toISOString(),
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
it("marks a rolled-back task distinctly from untouched pending work", () => {
|
|
283
|
+
// A rollback keeps the task's evidence (Task-2.1), so "pending with
|
|
284
|
+
// evidence" is the observable signature of work that was reopened.
|
|
285
|
+
const tasks = withProgress({
|
|
286
|
+
"Task-1": { status: "pending" },
|
|
287
|
+
"Task-2": { status: "pending", evidence: "wired the tool" },
|
|
288
|
+
}).tasks;
|
|
289
|
+
const rolled = tasks.find((t) => t.id === "Task-2")!;
|
|
290
|
+
const untouched = tasks.find((t) => t.id === "Task-1")!;
|
|
291
|
+
assert.equal(taskMarker(rolled, null), "↺");
|
|
292
|
+
assert.equal(taskMarker(untouched, null), "·");
|
|
293
|
+
// The current-task indicator still wins over the rollback marker.
|
|
294
|
+
assert.equal(taskMarker(rolled, "Task-2"), "▸");
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
it("shows the retained evidence on a rolled-back row in the tree view", () => {
|
|
298
|
+
const tree = renderDashboardTreeLines(
|
|
299
|
+
withProgress({
|
|
300
|
+
"Task-1": { status: "pending" },
|
|
301
|
+
"Task-2": { status: "pending", evidence: "wired the tool" },
|
|
302
|
+
}),
|
|
303
|
+
100,
|
|
304
|
+
).join("\n");
|
|
305
|
+
assert.match(tree, /↺ Task-2/, "the rolled-back row carries its own marker");
|
|
306
|
+
assert.match(tree, /wired the tool/, "the previous attempt's evidence is visible");
|
|
307
|
+
});
|
|
308
|
+
|
|
309
|
+
const terminal = (extra: { auditRounds?: number; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean }) => {
|
|
310
|
+
const base = withProgress({
|
|
311
|
+
"Task-1": { status: "complete", evidence: "done" },
|
|
312
|
+
"Task-2": { status: "complete", evidence: "done" },
|
|
313
|
+
"Task-3": { status: "complete", evidence: "done" },
|
|
314
|
+
"Task-3.1": { status: "complete", evidence: "done" },
|
|
315
|
+
"Task-3.2": { status: "complete", evidence: "done" },
|
|
316
|
+
});
|
|
317
|
+
return deriveDashboardModel("demo-run", base.tasks, base.checklist, { startedAt: base.startedAt, ...extra });
|
|
318
|
+
};
|
|
319
|
+
|
|
320
|
+
it("shows the undeterminable count instead of a completion tick", () => {
|
|
321
|
+
const owed = terminal({ auditRounds: 2, auditUndeterminable: ["VC-001", "VC-002"] });
|
|
322
|
+
const panel = renderDashboardLines(owed, 80);
|
|
323
|
+
assert.ok(panel.some((l) => /undeterminable/.test(l)), "panel reports undeterminable");
|
|
324
|
+
assert.ok(!panel.some((l) => /audit complete/.test(l)), "no false completion tick");
|
|
325
|
+
const tree = renderDashboardTreeLines(owed, 100).join("\n");
|
|
326
|
+
assert.match(tree, /undeterminable: VC-001, VC-002/);
|
|
327
|
+
assert.doesNotMatch(tree, /passed ✓/);
|
|
328
|
+
});
|
|
329
|
+
|
|
330
|
+
it("shows the retained evidence for rolled-back work in the COMPACT view too", () => {
|
|
331
|
+
// VC-006 requires the evidence in both views. The first implementation
|
|
332
|
+
// only added it to the tree, which the execution review caught.
|
|
333
|
+
const m = withProgress({
|
|
334
|
+
"Task-1": { status: "pending", evidence: "rewrote src/plan.ts" },
|
|
335
|
+
"Task-2": { status: "complete", evidence: "wired the tool" },
|
|
336
|
+
});
|
|
337
|
+
const panel = renderDashboardLines(m, 90).join("\n");
|
|
338
|
+
assert.match(panel, /\u21ba Task-1/, "compact panel marks the rolled-back task");
|
|
339
|
+
assert.match(panel, /rewrote src\/plan\.ts/, "compact panel shows the retained evidence");
|
|
340
|
+
const tree = renderDashboardTreeLines(m, 100).join("\n");
|
|
341
|
+
assert.match(tree, /rewrote src\/plan\.ts/, "tree view still shows it");
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
it("keeps every compact row inside the width budget with rollback rows present", () => {
|
|
345
|
+
const m = deriveDashboardModel(
|
|
346
|
+
"demo-run",
|
|
347
|
+
buildTaskView(parsePlanTasks(PLAN), {
|
|
348
|
+
"Task-1": { status: "pending", evidence: "a very long piece of evidence that must be clipped to fit the panel width" },
|
|
349
|
+
"Task-2": { status: "pending", evidence: "another long evidence string for the second rolled-back task" },
|
|
350
|
+
"Task-3": { status: "pending", evidence: "a third one that must be capped out of the panel" },
|
|
351
|
+
}),
|
|
352
|
+
CHECKS(),
|
|
353
|
+
{ startedAt: new Date().toISOString(), auditRounds: 1, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] },
|
|
354
|
+
);
|
|
355
|
+
for (const width of [46, 64, 80, 120]) {
|
|
356
|
+
for (const row of renderDashboardLines(m, width)) {
|
|
357
|
+
assert.ok(visibleWidth(row) <= width, `width ${width}: ${JSON.stringify(row)}`);
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
const rows = renderDashboardLines(m, 90).filter((l) => /\u21ba /.test(l));
|
|
361
|
+
assert.equal(rows.length, 2, "rollback rows are capped so the panel height stays bounded");
|
|
362
|
+
});
|
|
363
|
+
|
|
364
|
+
it("never shows the completion tick while a review round runs or checks are owed (v0.8)", () => {
|
|
365
|
+
// The v0.7 mis-cue: `audit complete ✓` rendered for the whole duration
|
|
366
|
+
// of a running round (auditRounds was pre-incremented, the model had no
|
|
367
|
+
// running field). Now the running/owed states render their round number.
|
|
368
|
+
const running = terminal({ auditRounds: 1, reviewRunning: true });
|
|
369
|
+
const runningPanel = renderDashboardLines(running, 80).join("\n");
|
|
370
|
+
assert.match(runningPanel, /review: round 2\/5 running/, "the running round is visible with its number");
|
|
371
|
+
assert.doesNotMatch(runningPanel, /audit complete/, "no tick while a round runs");
|
|
372
|
+
const runningTree = renderDashboardTreeLines(running, 100).join("\n");
|
|
373
|
+
assert.match(runningTree, /Execution review: round 2\/5 running/);
|
|
374
|
+
const owed = terminal({ auditRounds: 1 });
|
|
375
|
+
const owedPanel = renderDashboardLines(owed, 80).join("\n");
|
|
376
|
+
assert.doesNotMatch(owedPanel, /audit complete/, "no tick while checks are still owed");
|
|
377
|
+
assert.match(owedPanel, /review: round 2\/5 — verdict pending/, "the owed state shows the pending round");
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
it("shows the completion tick only when every check is really done", () => {
|
|
381
|
+
const settled = terminal({ auditRounds: 1 });
|
|
382
|
+
const model = { ...settled, checklist: settled.checklist.map((item) => ({ ...item, done: true })) };
|
|
383
|
+
assert.ok(renderDashboardLines(model, 80).some((l) => /audit complete/.test(l)));
|
|
384
|
+
});
|
|
385
|
+
|
|
386
|
+
it("still reports real failures ahead of undeterminable ones", () => {
|
|
387
|
+
const mixed = terminal({ auditRounds: 2, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] });
|
|
388
|
+
const tree = renderDashboardTreeLines(mixed, 100).join("\n");
|
|
389
|
+
assert.match(tree, /failed: VC-001/, "a real failure takes the tree audit line");
|
|
390
|
+
assert.doesNotMatch(tree, /undeterminable:/, "and suppresses the softer outcome there");
|
|
391
|
+
const panel = renderDashboardLines(mixed, 80);
|
|
392
|
+
assert.ok(panel.some((l) => /check\(s\) failed/.test(l)));
|
|
393
|
+
assert.ok(panel.some((l) => /\? VC-002/.test(l)), "the unreadable checks still get their own panel row");
|
|
394
|
+
});
|
|
395
|
+
|
|
396
|
+
it("keeps the width invariant with the new rows", () => {
|
|
397
|
+
const rows = renderDashboardLines(terminal({ auditRounds: 2, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] }), 46);
|
|
398
|
+
for (const row of rows) {
|
|
399
|
+
assert.ok(visibleWidth(row) <= 46, `row too wide: ${JSON.stringify(row)}`);
|
|
400
|
+
}
|
|
401
|
+
});
|
|
402
|
+
});
|
|
403
|
+
|
|
404
|
+
describe("findings visibility (v0.9)", () => {
|
|
405
|
+
const findings = [
|
|
406
|
+
{ id: "F-001", severity: "high", note: "union rollback missing", taskIds: ["Task-3"] },
|
|
407
|
+
{ id: "F-002", severity: "medium", note: "polish", taskIds: [] },
|
|
408
|
+
];
|
|
409
|
+
|
|
410
|
+
function withFindings(extra: { auditRounds?: number; reviewRunning?: boolean }) {
|
|
411
|
+
const tasks = buildTaskView(parsePlanTasks(PLAN), {});
|
|
412
|
+
const checklist = [
|
|
413
|
+
{ id: "VC-001", text: "`VC-001` covers `Task-1`; pass condition: x", done: false },
|
|
414
|
+
{ id: "VC-002", text: "`VC-002` covers `Task-3`; pass condition: y", done: false },
|
|
415
|
+
];
|
|
416
|
+
return deriveDashboardModel("demo-run", tasks, checklist, {
|
|
417
|
+
startedAt: new Date().toISOString(),
|
|
418
|
+
findings,
|
|
419
|
+
auditRounds: extra.auditRounds ?? null,
|
|
420
|
+
reviewRunning: extra.reviewRunning ?? false,
|
|
421
|
+
});
|
|
422
|
+
}
|
|
423
|
+
|
|
424
|
+
it("summary line shows review round/5 and the high count", () => {
|
|
425
|
+
const line = formatDashboardSummaryLine(withFindings({ auditRounds: 2 }));
|
|
426
|
+
assert.match(line, /review r2\/5/);
|
|
427
|
+
assert.match(line, /1 high/);
|
|
428
|
+
});
|
|
429
|
+
|
|
430
|
+
it("summary line omits the high token when only non-high findings remain", () => {
|
|
431
|
+
const tasks = buildTaskView(parsePlanTasks(PLAN), {});
|
|
432
|
+
const m = deriveDashboardModel("demo-run", tasks, [], {
|
|
433
|
+
startedAt: new Date().toISOString(),
|
|
434
|
+
findings: [{ id: "F-002", severity: "medium", note: "polish", taskIds: [] }],
|
|
435
|
+
auditRounds: 3,
|
|
436
|
+
});
|
|
437
|
+
const line = formatDashboardSummaryLine(m);
|
|
438
|
+
assert.match(line, /review r3\/5/);
|
|
439
|
+
assert.doesNotMatch(line, /high/);
|
|
440
|
+
});
|
|
441
|
+
|
|
442
|
+
it("compact panel renders the high-findings line in BOTH phases", () => {
|
|
443
|
+
// Non-terminal phase (the executor is repairing): the line must show.
|
|
444
|
+
const repairing = renderDashboardLines(withFindings({ auditRounds: 1 }), 80);
|
|
445
|
+
assert.ok(repairing.some((l) => /⚠.*high: F-001/.test(l)), "high findings visible while repairing");
|
|
446
|
+
|
|
447
|
+
// Terminal phase: still visible alongside the round counter.
|
|
448
|
+
const everyId = Object.fromEntries(
|
|
449
|
+
flattenTaskViews(buildTaskView(parsePlanTasks(PLAN), {})).map((t) => [t.id, { status: "complete" as const }]),
|
|
450
|
+
);
|
|
451
|
+
const allDone = buildTaskView(parsePlanTasks(PLAN), everyId);
|
|
452
|
+
const terminal = deriveDashboardModel("demo-run", allDone, [], {
|
|
453
|
+
startedAt: new Date().toISOString(),
|
|
454
|
+
findings,
|
|
455
|
+
auditRounds: 1,
|
|
456
|
+
});
|
|
457
|
+
const lines = renderDashboardLines(terminal, 80);
|
|
458
|
+
assert.ok(lines.some((l) => /⚠.*high: F-001/.test(l)));
|
|
459
|
+
assert.ok(lines.some((l) => /high finding\(s\) unresolved/.test(l)));
|
|
460
|
+
});
|
|
461
|
+
|
|
462
|
+
it("tree view lists findings with severity and mapping, and the verdict line names unresolved highs", () => {
|
|
463
|
+
const lines = renderDashboardTreeLines(withFindings({ auditRounds: 2 }), 100);
|
|
464
|
+
assert.ok(lines.some((l) => /⚠ F-001 \(high, Task-3\): union rollback missing/.test(l)));
|
|
465
|
+
assert.ok(lines.some((l) => /· F-002 \(medium\): polish/.test(l)));
|
|
466
|
+
assert.ok(lines.some((l) => /high findings unresolved: F-001/.test(l)));
|
|
467
|
+
});
|
|
468
|
+
});
|