pi-plans 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +40 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +13 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +157 -56
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +56 -10
- package/src/exec.ts +614 -126
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +13 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +39 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +18 -5
- package/tests/auditor.test.ts +116 -17
- package/tests/dashboard.test.ts +135 -1
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +65 -0
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
package/tests/dashboard.test.ts
CHANGED
|
@@ -134,7 +134,7 @@ describe("expanded tree rendering", () => {
|
|
|
134
134
|
m.auditRounds = 2;
|
|
135
135
|
m.auditFailed = ["VC-001"];
|
|
136
136
|
const lines = renderDashboardTreeLines(m, 100);
|
|
137
|
-
assert.ok(lines.some((line) => line.includes("
|
|
137
|
+
assert.ok(lines.some((line) => line.includes("Execution review: round 2")));
|
|
138
138
|
});
|
|
139
139
|
});
|
|
140
140
|
|
|
@@ -266,3 +266,137 @@ describe("width invariant (TUI crash regression)", () => {
|
|
|
266
266
|
}
|
|
267
267
|
});
|
|
268
268
|
});
|
|
269
|
+
|
|
270
|
+
describe("rolled-back task rendering", () => {
|
|
271
|
+
const CHECKS = () => [
|
|
272
|
+
{ id: "VC-001", text: "`VC-001` covers `Task-1`; pass condition: x", done: false },
|
|
273
|
+
{ id: "VC-002", text: "`VC-002` covers `Task-3`; pass condition: y", done: false },
|
|
274
|
+
];
|
|
275
|
+
|
|
276
|
+
function withProgress(progress: Record<string, { status: "complete" | "skipped" | "pending"; evidence?: string }>) {
|
|
277
|
+
return deriveDashboardModel("demo-run", buildTaskView(parsePlanTasks(PLAN), progress), CHECKS(), {
|
|
278
|
+
startedAt: new Date().toISOString(),
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
it("marks a rolled-back task distinctly from untouched pending work", () => {
|
|
283
|
+
// A rollback keeps the task's evidence (Task-2.1), so "pending with
|
|
284
|
+
// evidence" is the observable signature of work that was reopened.
|
|
285
|
+
const tasks = withProgress({
|
|
286
|
+
"Task-1": { status: "pending" },
|
|
287
|
+
"Task-2": { status: "pending", evidence: "wired the tool" },
|
|
288
|
+
}).tasks;
|
|
289
|
+
const rolled = tasks.find((t) => t.id === "Task-2")!;
|
|
290
|
+
const untouched = tasks.find((t) => t.id === "Task-1")!;
|
|
291
|
+
assert.equal(taskMarker(rolled, null), "↺");
|
|
292
|
+
assert.equal(taskMarker(untouched, null), "·");
|
|
293
|
+
// The current-task indicator still wins over the rollback marker.
|
|
294
|
+
assert.equal(taskMarker(rolled, "Task-2"), "▸");
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
it("shows the retained evidence on a rolled-back row in the tree view", () => {
|
|
298
|
+
const tree = renderDashboardTreeLines(
|
|
299
|
+
withProgress({
|
|
300
|
+
"Task-1": { status: "pending" },
|
|
301
|
+
"Task-2": { status: "pending", evidence: "wired the tool" },
|
|
302
|
+
}),
|
|
303
|
+
100,
|
|
304
|
+
).join("\n");
|
|
305
|
+
assert.match(tree, /↺ Task-2/, "the rolled-back row carries its own marker");
|
|
306
|
+
assert.match(tree, /wired the tool/, "the previous attempt's evidence is visible");
|
|
307
|
+
});
|
|
308
|
+
|
|
309
|
+
const terminal = (extra: { auditRounds?: number; auditFailed?: string[]; auditUndeterminable?: string[]; reviewRunning?: boolean }) => {
|
|
310
|
+
const base = withProgress({
|
|
311
|
+
"Task-1": { status: "complete", evidence: "done" },
|
|
312
|
+
"Task-2": { status: "complete", evidence: "done" },
|
|
313
|
+
"Task-3": { status: "complete", evidence: "done" },
|
|
314
|
+
"Task-3.1": { status: "complete", evidence: "done" },
|
|
315
|
+
"Task-3.2": { status: "complete", evidence: "done" },
|
|
316
|
+
});
|
|
317
|
+
return deriveDashboardModel("demo-run", base.tasks, base.checklist, { startedAt: base.startedAt, ...extra });
|
|
318
|
+
};
|
|
319
|
+
|
|
320
|
+
it("shows the undeterminable count instead of a completion tick", () => {
|
|
321
|
+
const owed = terminal({ auditRounds: 2, auditUndeterminable: ["VC-001", "VC-002"] });
|
|
322
|
+
const panel = renderDashboardLines(owed, 80);
|
|
323
|
+
assert.ok(panel.some((l) => /undeterminable/.test(l)), "panel reports undeterminable");
|
|
324
|
+
assert.ok(!panel.some((l) => /audit complete/.test(l)), "no false completion tick");
|
|
325
|
+
const tree = renderDashboardTreeLines(owed, 100).join("\n");
|
|
326
|
+
assert.match(tree, /undeterminable: VC-001, VC-002/);
|
|
327
|
+
assert.doesNotMatch(tree, /passed ✓/);
|
|
328
|
+
});
|
|
329
|
+
|
|
330
|
+
it("shows the retained evidence for rolled-back work in the COMPACT view too", () => {
|
|
331
|
+
// VC-006 requires the evidence in both views. The first implementation
|
|
332
|
+
// only added it to the tree, which the execution review caught.
|
|
333
|
+
const m = withProgress({
|
|
334
|
+
"Task-1": { status: "pending", evidence: "rewrote src/plan.ts" },
|
|
335
|
+
"Task-2": { status: "complete", evidence: "wired the tool" },
|
|
336
|
+
});
|
|
337
|
+
const panel = renderDashboardLines(m, 90).join("\n");
|
|
338
|
+
assert.match(panel, /\u21ba Task-1/, "compact panel marks the rolled-back task");
|
|
339
|
+
assert.match(panel, /rewrote src\/plan\.ts/, "compact panel shows the retained evidence");
|
|
340
|
+
const tree = renderDashboardTreeLines(m, 100).join("\n");
|
|
341
|
+
assert.match(tree, /rewrote src\/plan\.ts/, "tree view still shows it");
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
it("keeps every compact row inside the width budget with rollback rows present", () => {
|
|
345
|
+
const m = deriveDashboardModel(
|
|
346
|
+
"demo-run",
|
|
347
|
+
buildTaskView(parsePlanTasks(PLAN), {
|
|
348
|
+
"Task-1": { status: "pending", evidence: "a very long piece of evidence that must be clipped to fit the panel width" },
|
|
349
|
+
"Task-2": { status: "pending", evidence: "another long evidence string for the second rolled-back task" },
|
|
350
|
+
"Task-3": { status: "pending", evidence: "a third one that must be capped out of the panel" },
|
|
351
|
+
}),
|
|
352
|
+
CHECKS(),
|
|
353
|
+
{ startedAt: new Date().toISOString(), auditRounds: 1, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] },
|
|
354
|
+
);
|
|
355
|
+
for (const width of [46, 64, 80, 120]) {
|
|
356
|
+
for (const row of renderDashboardLines(m, width)) {
|
|
357
|
+
assert.ok(visibleWidth(row) <= width, `width ${width}: ${JSON.stringify(row)}`);
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
const rows = renderDashboardLines(m, 90).filter((l) => /\u21ba /.test(l));
|
|
361
|
+
assert.equal(rows.length, 2, "rollback rows are capped so the panel height stays bounded");
|
|
362
|
+
});
|
|
363
|
+
|
|
364
|
+
it("never shows the completion tick while a review round runs or checks are owed (v0.8)", () => {
|
|
365
|
+
// The v0.7 mis-cue: `audit complete ✓` rendered for the whole duration
|
|
366
|
+
// of a running round (auditRounds was pre-incremented, the model had no
|
|
367
|
+
// running field). Now the running/owed states render their round number.
|
|
368
|
+
const running = terminal({ auditRounds: 1, reviewRunning: true });
|
|
369
|
+
const runningPanel = renderDashboardLines(running, 80).join("\n");
|
|
370
|
+
assert.match(runningPanel, /review: round 2\/5 running/, "the running round is visible with its number");
|
|
371
|
+
assert.doesNotMatch(runningPanel, /audit complete/, "no tick while a round runs");
|
|
372
|
+
const runningTree = renderDashboardTreeLines(running, 100).join("\n");
|
|
373
|
+
assert.match(runningTree, /Execution review: round 2\/5 running/);
|
|
374
|
+
const owed = terminal({ auditRounds: 1 });
|
|
375
|
+
const owedPanel = renderDashboardLines(owed, 80).join("\n");
|
|
376
|
+
assert.doesNotMatch(owedPanel, /audit complete/, "no tick while checks are still owed");
|
|
377
|
+
assert.match(owedPanel, /review: round 2\/5 — verdict pending/, "the owed state shows the pending round");
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
it("shows the completion tick only when every check is really done", () => {
|
|
381
|
+
const settled = terminal({ auditRounds: 1 });
|
|
382
|
+
const model = { ...settled, checklist: settled.checklist.map((item) => ({ ...item, done: true })) };
|
|
383
|
+
assert.ok(renderDashboardLines(model, 80).some((l) => /audit complete/.test(l)));
|
|
384
|
+
});
|
|
385
|
+
|
|
386
|
+
it("still reports real failures ahead of undeterminable ones", () => {
|
|
387
|
+
const mixed = terminal({ auditRounds: 2, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] });
|
|
388
|
+
const tree = renderDashboardTreeLines(mixed, 100).join("\n");
|
|
389
|
+
assert.match(tree, /failed: VC-001/, "a real failure takes the tree audit line");
|
|
390
|
+
assert.doesNotMatch(tree, /undeterminable:/, "and suppresses the softer outcome there");
|
|
391
|
+
const panel = renderDashboardLines(mixed, 80);
|
|
392
|
+
assert.ok(panel.some((l) => /check\(s\) failed/.test(l)));
|
|
393
|
+
assert.ok(panel.some((l) => /\? VC-002/.test(l)), "the unreadable checks still get their own panel row");
|
|
394
|
+
});
|
|
395
|
+
|
|
396
|
+
it("keeps the width invariant with the new rows", () => {
|
|
397
|
+
const rows = renderDashboardLines(terminal({ auditRounds: 2, auditFailed: ["VC-001"], auditUndeterminable: ["VC-002"] }), 46);
|
|
398
|
+
for (const row of rows) {
|
|
399
|
+
assert.ok(visibleWidth(row) <= 46, `row too wide: ${JSON.stringify(row)}`);
|
|
400
|
+
}
|
|
401
|
+
});
|
|
402
|
+
});
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Execution-review loop tests (v0.8): detached rounds, abort/identity
|
|
3
|
+
* lifecycle, the fingerprint guard, budget accounting (commit-only), round
|
|
4
|
+
* report persistence, the cap pause, and resume self-heal.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import * as assert from "node:assert/strict";
|
|
8
|
+
import * as fs from "node:fs";
|
|
9
|
+
import * as os from "node:os";
|
|
10
|
+
import * as path from "node:path";
|
|
11
|
+
import { after, before, describe, it } from "node:test";
|
|
12
|
+
import {
|
|
13
|
+
__awaitReviewRoundForTests,
|
|
14
|
+
__setAuditRunnerForTests,
|
|
15
|
+
getExecution,
|
|
16
|
+
persistTaskProgress,
|
|
17
|
+
restoreFromSession,
|
|
18
|
+
startExecution,
|
|
19
|
+
stopExecution,
|
|
20
|
+
} from "../src/exec.ts";
|
|
21
|
+
import { setMessagingApi } from "../src/messaging.ts";
|
|
22
|
+
import { applyTaskUpdate } from "../src/task-tool.ts";
|
|
23
|
+
import { REVIEW_MAX_ROUNDS } from "../src/auditor.ts";
|
|
24
|
+
import { getRun, initState, startRun } from "../src/state.ts";
|
|
25
|
+
import { createCheckpoint, loadCheckpoint, mutateCheckpoint, applyExecutionApproved, applyExecutionProgress, applyPlanWritten, planIdentityOf } from "../src/workflow-state.ts";
|
|
26
|
+
|
|
27
|
+
const PLAN = `# PLAN_v1 - review-loop fixture
|
|
28
|
+
|
|
29
|
+
## Tasks
|
|
30
|
+
|
|
31
|
+
- Task-1: engine — files: lib/engine.js; wave: 1
|
|
32
|
+
- Task-2: report — files: lib/report.js; wave: 1
|
|
33
|
+
|
|
34
|
+
## Verification Checks
|
|
35
|
+
|
|
36
|
+
- [ ] \`VC-001\` covers \`Task-1\`; pass condition: engine works; evidence: tests; metric: green.
|
|
37
|
+
- [ ] \`VC-002\` covers \`Task-2\`; pass condition: report written; evidence: file; metric: green.
|
|
38
|
+
`;
|
|
39
|
+
|
|
40
|
+
let root = "";
|
|
41
|
+
let counter = 0;
|
|
42
|
+
|
|
43
|
+
before(() => {
|
|
44
|
+
root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-review-loop-"));
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
after(() => {
|
|
48
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
49
|
+
__setAuditRunnerForTests(null);
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
function freshWorkdir(): { workdir: string; planPath: string; runId: string } {
|
|
53
|
+
counter += 1;
|
|
54
|
+
const workdir = path.join(root, `repo-${counter}`);
|
|
55
|
+
fs.mkdirSync(workdir, { recursive: true });
|
|
56
|
+
fs.mkdirSync(path.join(workdir, "lib"), { recursive: true });
|
|
57
|
+
// Real covered files so the round fingerprint tracks their mtimes.
|
|
58
|
+
fs.writeFileSync(path.join(workdir, "lib", "engine.js"), "export const engine = 1;\n", "utf8");
|
|
59
|
+
fs.writeFileSync(path.join(workdir, "lib", "report.js"), "export const report = 1;\n", "utf8");
|
|
60
|
+
initState(workdir);
|
|
61
|
+
const { run } = startRun(workdir, { topic: `r${counter}`, skill: "plan-small", requestText: "demo" });
|
|
62
|
+
createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
|
|
63
|
+
const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
|
|
64
|
+
fs.mkdirSync(run.artifact_dir, { recursive: true });
|
|
65
|
+
fs.writeFileSync(planPath, PLAN, "utf8");
|
|
66
|
+
mutateCheckpoint(workdir, run.run_id, (cp) =>
|
|
67
|
+
applyExecutionApproved(
|
|
68
|
+
applyPlanWritten({ ...cp, nextAction: "accept-execute" }, planIdentityOf(planPath, 1)),
|
|
69
|
+
{ plan: planIdentityOf(planPath, 1), worktree: workdir, headAtApproval: null, approvedAt: cp.updatedAt },
|
|
70
|
+
),
|
|
71
|
+
);
|
|
72
|
+
return { workdir, planPath, runId: run.run_id };
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function makeCtx(workdir: string, mode: "print" | "tui" = "print", customOpens?: { count: number }) {
|
|
76
|
+
const entries: Array<{ customType: string; data?: unknown; content?: string }> = [];
|
|
77
|
+
const ctx = {
|
|
78
|
+
cwd: workdir,
|
|
79
|
+
sessionManager: {},
|
|
80
|
+
hasUI: true,
|
|
81
|
+
mode,
|
|
82
|
+
entries,
|
|
83
|
+
ui: {
|
|
84
|
+
notify: () => {},
|
|
85
|
+
setStatus: () => {},
|
|
86
|
+
setWidget: () => {},
|
|
87
|
+
theme: { fg: (_c: string, t: string) => t, bold: (t: string) => t },
|
|
88
|
+
// Minimal overlay host: counts ui.custom opens (one per controller).
|
|
89
|
+
custom: () => {
|
|
90
|
+
if (customOpens) customOpens.count += 1;
|
|
91
|
+
return Promise.resolve();
|
|
92
|
+
},
|
|
93
|
+
},
|
|
94
|
+
isIdle: () => true,
|
|
95
|
+
hasPendingMessages: () => false,
|
|
96
|
+
} as never;
|
|
97
|
+
setMessagingApi({
|
|
98
|
+
appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }),
|
|
99
|
+
sendMessage: (message: { customType: string; content: string }) => entries.push({ customType: message.customType, content: message.content }),
|
|
100
|
+
} as never);
|
|
101
|
+
return ctx;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
async function startTerminal(planPath: string, workdir: string) {
|
|
105
|
+
const ctx = makeCtx(workdir);
|
|
106
|
+
await startExecution(ctx, { planPath, planTasks: (await import("../src/plan.ts")).parsePlanTasks(PLAN), items: (await import("../src/plan.ts")).parseChecklist(PLAN) });
|
|
107
|
+
for (const id of ["Task-1", "Task-2"]) {
|
|
108
|
+
applyTaskUpdate(getExecution()!.tasks, id, "complete", `${id} evidence`);
|
|
109
|
+
persistTaskProgress(ctx);
|
|
110
|
+
}
|
|
111
|
+
return ctx;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** A runner whose rounds the test resolves by hand. */
|
|
115
|
+
function controlledRunner() {
|
|
116
|
+
const pending: Array<(value: { round: number; passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null) => void> = [];
|
|
117
|
+
let calls = 0;
|
|
118
|
+
const runner = async () => {
|
|
119
|
+
calls += 1;
|
|
120
|
+
return new Promise<{ round: number; passed: string[]; failed: string[]; undeterminable: string[]; report: string } | null>((resolve) => pending.push(resolve));
|
|
121
|
+
};
|
|
122
|
+
const resolveRound = (value: Parameters<typeof pending[0]>[0]) => {
|
|
123
|
+
const resolve = pending.shift();
|
|
124
|
+
if (resolve) resolve(value);
|
|
125
|
+
};
|
|
126
|
+
/** Resolve every still-pending round (teardown: the identity guard discards them). */
|
|
127
|
+
const drainAll = (value: Parameters<typeof pending[0]>[0] = null) => {
|
|
128
|
+
while (pending.length > 0) pending.shift()!(value);
|
|
129
|
+
};
|
|
130
|
+
return { runner, resolveRound, drainAll, calls: () => calls };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Let the engine chain advance one step without awaiting its completion. */
|
|
134
|
+
const tick = async (): Promise<void> => {
|
|
135
|
+
await new Promise<void>((resolve) => setImmediate(resolve));
|
|
136
|
+
await new Promise<void>((resolve) => setImmediate(resolve));
|
|
137
|
+
};
|
|
138
|
+
|
|
139
|
+
describe("execution-review loop (v0.8)", () => {
|
|
140
|
+
it("tui/rpc detach: the settle returns before the round resolves, status becomes verifying", async () => {
|
|
141
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
142
|
+
await startTerminal(planPath, workdir);
|
|
143
|
+
const ctl = controlledRunner();
|
|
144
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
145
|
+
const ctxTui = makeCtx(workdir, "tui");
|
|
146
|
+
await restoreFromSession(ctxTui, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
147
|
+
// The restore returned while the round is STILL running: in-flight marker
|
|
148
|
+
// present, run status moved to verifying, no outcome committed yet.
|
|
149
|
+
assert.ok(getExecution()!.review.inFlight, "the round is in flight after the settle returned");
|
|
150
|
+
assert.equal(getExecution()!.audit.rounds, 0, "no budget spent yet");
|
|
151
|
+
assert.equal(getRun(workdir, runId)?.status, "verifying", "run status enters verifying");
|
|
152
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
|
|
153
|
+
await __awaitReviewRoundForTests();
|
|
154
|
+
assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed", "the round completes the run after resolution");
|
|
155
|
+
__setAuditRunnerForTests(null);
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
it("print/json keep the inline await: the settle returns only after the round commits", async () => {
|
|
159
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
160
|
+
await startTerminal(planPath, workdir);
|
|
161
|
+
const ctl = controlledRunner();
|
|
162
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
163
|
+
const ctxPrint = makeCtx(workdir, "print");
|
|
164
|
+
const restoring = restoreFromSession(ctxPrint, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
165
|
+
await tick(); // the inline round spawns
|
|
166
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "all pass" });
|
|
167
|
+
await restoring;
|
|
168
|
+
// Inline: by the time restoreFromSession returned, the round committed —
|
|
169
|
+
// and a passing round completes (clearing execution state).
|
|
170
|
+
assert.equal(getExecution(), null, "the inline round committed before the settle returned");
|
|
171
|
+
assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
|
|
172
|
+
assert.equal(ctl.calls(), 1);
|
|
173
|
+
__setAuditRunnerForTests(null);
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
it("a cancelled round burns no budget, sends no wake, and pauses nothing", async () => {
|
|
177
|
+
const { workdir, planPath } = freshWorkdir();
|
|
178
|
+
const ctx = await startTerminal(planPath, workdir);
|
|
179
|
+
__setAuditRunnerForTests(async () => ({ cancelled: true }) as never);
|
|
180
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
181
|
+
const ex = getExecution()!;
|
|
182
|
+
assert.equal(ex.audit.rounds, 0, "cancelled rounds burn no budget");
|
|
183
|
+
assert.equal(ex.stall.paused, false, "cancelled rounds do not pause");
|
|
184
|
+
assert.equal(ctx.entries.filter((e) => String(e.customType).startsWith("pi-plans-audit-")).length, 0, "no wake, no report message");
|
|
185
|
+
await stopExecution(ctx, "teardown");
|
|
186
|
+
});
|
|
187
|
+
|
|
188
|
+
it("fingerprint change discards the attempt, re-runs without budget burn, and both report files survive", async () => {
|
|
189
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
190
|
+
const ctx = await startTerminal(planPath, workdir);
|
|
191
|
+
const ctl = controlledRunner();
|
|
192
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
193
|
+
const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
194
|
+
await tick(); // the inline round spawns
|
|
195
|
+
// Mutate a covered file while the round is in flight (real mutation path).
|
|
196
|
+
const engine = path.join(workdir, "lib", "engine.js");
|
|
197
|
+
fs.writeFileSync(engine, "export const engine = 2;\n", "utf8");
|
|
198
|
+
const later = new Date(Date.now() + 10_000);
|
|
199
|
+
fs.utimesSync(engine, later, later);
|
|
200
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "stale verdict" });
|
|
201
|
+
await tick();
|
|
202
|
+
// Attempt 1 discarded (report on disk, marked), attempt 2 committed the pass.
|
|
203
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "fresh verdict" });
|
|
204
|
+
await restoring;
|
|
205
|
+
await __awaitReviewRoundForTests();
|
|
206
|
+
// Attempt 1 discarded (report on disk, marked), attempt 2 committed the pass.
|
|
207
|
+
const ex = getExecution();
|
|
208
|
+
assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "completed");
|
|
209
|
+
const runDir = path.join(workdir, ".git", "pi-plans", "runs", runId);
|
|
210
|
+
const attempt1 = path.join(runDir, "execution-review", "round-1-attempt-1.md");
|
|
211
|
+
const attempt2 = path.join(runDir, "execution-review", "round-1-attempt-2.md");
|
|
212
|
+
assert.ok(fs.existsSync(attempt1), "the discarded attempt's report survives");
|
|
213
|
+
assert.match(fs.readFileSync(attempt1, "utf8"), /outcome: discarded/);
|
|
214
|
+
assert.match(fs.readFileSync(attempt1, "utf8"), /discarded: fingerprint changed/);
|
|
215
|
+
assert.ok(fs.existsSync(attempt2), "the re-run's report exists beside it");
|
|
216
|
+
assert.match(fs.readFileSync(attempt2, "utf8"), /outcome: passed/);
|
|
217
|
+
assert.equal(ctl.calls(), 2, "one discard plus one re-run");
|
|
218
|
+
assert.ok(!ex, "execution completed");
|
|
219
|
+
__setAuditRunnerForTests(null);
|
|
220
|
+
});
|
|
221
|
+
|
|
222
|
+
it("two consecutive discards commit as a budget-counting undeterminable round", async () => {
|
|
223
|
+
const { workdir, planPath } = freshWorkdir();
|
|
224
|
+
const ctx = await startTerminal(planPath, workdir);
|
|
225
|
+
const ctl = controlledRunner();
|
|
226
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
227
|
+
const engine = path.join(workdir, "lib", "engine.js");
|
|
228
|
+
const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
229
|
+
await tick();
|
|
230
|
+
const discard = () => {
|
|
231
|
+
fs.writeFileSync(engine, `export const engine = ${Math.random()};\n`, "utf8");
|
|
232
|
+
const later = new Date(Date.now() + 60_000);
|
|
233
|
+
fs.utimesSync(engine, later, later);
|
|
234
|
+
};
|
|
235
|
+
discard();
|
|
236
|
+
ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale 1" });
|
|
237
|
+
await tick();
|
|
238
|
+
assert.equal(getExecution()!.audit.rounds, 0, "the first discard burns nothing");
|
|
239
|
+
discard();
|
|
240
|
+
ctl.resolveRound({ round: 1, passed: [], failed: [], undeterminable: ["VC-001", "VC-002"], report: "stale 2" });
|
|
241
|
+
await tick();
|
|
242
|
+
assert.equal(getExecution()!.audit.rounds, 1, "the second consecutive discard commits as undeterminable (budget spent)");
|
|
243
|
+
assert.deepEqual(getExecution()!.audit.undeterminable, ["VC-001", "VC-002"]);
|
|
244
|
+
await stopExecution(ctx, "teardown");
|
|
245
|
+
ctl.drainAll();
|
|
246
|
+
await restoring;
|
|
247
|
+
await __awaitReviewRoundForTests();
|
|
248
|
+
__setAuditRunnerForTests(null);
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
it("a failed round rolls back, returns the run to executing, and wakes exactly once", async () => {
|
|
252
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
253
|
+
const ctx = await startTerminal(planPath, workdir);
|
|
254
|
+
const ctl = controlledRunner();
|
|
255
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
256
|
+
const restoring = restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
257
|
+
await tick();
|
|
258
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001"], failed: ["VC-002"], undeterminable: [], report: "VC-002 is not satisfied" });
|
|
259
|
+
await restoring;
|
|
260
|
+
await __awaitReviewRoundForTests();
|
|
261
|
+
const ex = getExecution()!;
|
|
262
|
+
assert.equal(getRun(workdir, runId)?.status, "executing", "rollback returns the run to executing for repair");
|
|
263
|
+
assert.equal(ex.tasks.find((t) => t.id === "Task-2")?.status, "pending", "the failed check's task rolled back");
|
|
264
|
+
const wakes = ctx.entries.filter((e) => e.customType === "pi-plans-audit-failed");
|
|
265
|
+
assert.equal(wakes.length, 1, "exactly one wake per committed failed outcome");
|
|
266
|
+
assert.equal(loadCheckpoint(workdir, runId).checkpoint.phase, "executing", "checkpoint phase stays executing across the round");
|
|
267
|
+
await stopExecution(ctx, "teardown");
|
|
268
|
+
__setAuditRunnerForTests(null);
|
|
269
|
+
});
|
|
270
|
+
|
|
271
|
+
it("a second restore aborts the previous in-flight round; never two live rounds on one run", async () => {
|
|
272
|
+
const { workdir, planPath } = freshWorkdir();
|
|
273
|
+
await startTerminal(planPath, workdir);
|
|
274
|
+
const ctl = controlledRunner();
|
|
275
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
276
|
+
const ctx = makeCtx(workdir, "tui");
|
|
277
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
278
|
+
const firstAttempt = getExecution()!.review.attempts;
|
|
279
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
280
|
+
const after = getExecution()!;
|
|
281
|
+
assert.ok(after.review.inFlight, "the second resume started its own round");
|
|
282
|
+
assert.equal(after.review.attempts, 1, "the replacement identity starts its own attempt 1 (the old round was aborted)");
|
|
283
|
+
assert.notEqual(after.review.inFlight.controller, undefined, "exactly one live round — its abort lifecycle is owned");
|
|
284
|
+
// The aborted first round resolves late against a replaced identity: it
|
|
285
|
+
// must be discarded silently (no extra wake, no budget charge).
|
|
286
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "late stale round" });
|
|
287
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "fresh round" });
|
|
288
|
+
await __awaitReviewRoundForTests();
|
|
289
|
+
assert.equal(ctx.entries.filter((e) => e.customType === "pi-plans-complete").length, 1, "the run completed exactly once");
|
|
290
|
+
__setAuditRunnerForTests(null);
|
|
291
|
+
});
|
|
292
|
+
|
|
293
|
+
it("every attempt opens a fresh overlay; the reopen shortcut is inert with no in-flight round", async () => {
|
|
294
|
+
const { workdir, planPath } = freshWorkdir();
|
|
295
|
+
await startTerminal(planPath, workdir);
|
|
296
|
+
const ctl = controlledRunner();
|
|
297
|
+
__setAuditRunnerForTests(ctl.runner);
|
|
298
|
+
const opens = { count: 0 };
|
|
299
|
+
const ctx = makeCtx(workdir, "tui", opens);
|
|
300
|
+
await restoreFromSession(ctx, [{ type: "custom", customType: "pi-plans-exec", data: getExecution() }]);
|
|
301
|
+
assert.equal(opens.count, 1, "the round opened its overlay before spawning");
|
|
302
|
+
// ESC closed the (one-shot) controller; the reopen shortcut rebuilds it
|
|
303
|
+
// from engine-held lane state while the round is still in flight.
|
|
304
|
+
const { reopenReviewOverlay } = await import("../src/exec.ts");
|
|
305
|
+
reopenReviewOverlay(ctx);
|
|
306
|
+
assert.equal(opens.count, 2, "reopen builds a fresh controller for the same round");
|
|
307
|
+
ctl.resolveRound({ round: 1, passed: ["VC-001", "VC-002"], failed: [], undeterminable: [], report: "done" });
|
|
308
|
+
await __awaitReviewRoundForTests();
|
|
309
|
+
// No round in flight → the shortcut is inert.
|
|
310
|
+
reopenReviewOverlay(ctx);
|
|
311
|
+
assert.equal(opens.count, 2, "reopen is inert with no in-flight round");
|
|
312
|
+
__setAuditRunnerForTests(null);
|
|
313
|
+
});
|
|
314
|
+
|
|
315
|
+
it("a legacy cap pause (old prefix) survives restore paused at its round count", async () => {
|
|
316
|
+
const { workdir, planPath, runId } = freshWorkdir();
|
|
317
|
+
await startTerminal(planPath, workdir);
|
|
318
|
+
// Simulate a checkpoint paused by a v0.7 build ("completion audit
|
|
319
|
+
// exhausted 3 rounds") — the dual-matched prefix must keep it paused.
|
|
320
|
+
mutateCheckpoint(workdir, runId, (cp) =>
|
|
321
|
+
applyExecutionProgress(cp, { audit: { rounds: 3, lastResult: "VC-001" }, pausedReason: "completion audit exhausted 3 rounds (failed: VC-001)." }),
|
|
322
|
+
);
|
|
323
|
+
const load = await import("../src/exec.ts");
|
|
324
|
+
const result = load.loadExecutionFromCheckpoint(makeCtx(workdir), runId);
|
|
325
|
+
assert.equal(result.status, "loaded", `checkpoint loads (${result.status})`);
|
|
326
|
+
const ex = getExecution()!;
|
|
327
|
+
assert.equal(ex.stall.paused, true, "a legacy cap pause stays paused across restore");
|
|
328
|
+
assert.equal(ex.audit.rounds, 3, "the legacy round count survives — restore grants no budget");
|
|
329
|
+
await stopExecution(makeCtx(workdir), "teardown");
|
|
330
|
+
});
|
|
331
|
+
});
|