pi-plans 0.7.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +5 -12
- package/README.md +5 -5
- package/agents/execution-reviewer.md +40 -0
- package/index.ts +13 -23
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +13 -7
- package/references/plan-artifact-template.md +11 -1
- package/references/state-and-config.md +3 -3
- package/scripts/validate.ts +20 -3
- package/src/auditor.ts +157 -56
- package/src/code-graph/commands.ts +6 -1
- package/src/dashboard.ts +56 -10
- package/src/exec.ts +614 -126
- package/src/plan.ts +1 -1
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +13 -3
- package/src/resume.ts +5 -1
- package/src/staleness.ts +53 -0
- package/src/state.ts +1 -0
- package/src/task-tool.ts +1 -1
- package/src/tasks.ts +39 -5
- package/src/ui-language.ts +4 -0
- package/src/workflow-state.ts +18 -5
- package/tests/auditor.test.ts +116 -17
- package/tests/dashboard.test.ts +135 -1
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +198 -44
- package/tests/extension-load.test.ts +1 -1
- package/tests/resume-lifecycle.test.ts +5 -1
- package/tests/resume.test.ts +6 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +4 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/workflow-state.test.ts +65 -0
- package/tools/execute-plan.ts +12 -5
- package/tools/plans.ts +1 -1
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
/** Tests for the task tree primitives that the execution review drives:
|
|
2
|
+
* rollback retention, persistence completeness, and done-flag invalidation.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
import * as assert from "node:assert/strict";
|
|
6
|
+
import { describe, it } from "node:test";
|
|
7
|
+
import type { CheckItem } from "../src/plan.ts";
|
|
8
|
+
import {
|
|
9
|
+
auditRollbackSet,
|
|
10
|
+
buildTaskView,
|
|
11
|
+
invalidateChecksForRolledBackTasks,
|
|
12
|
+
taskProgressMap,
|
|
13
|
+
type TaskProgressMap,
|
|
14
|
+
} from "../src/tasks.ts";
|
|
15
|
+
import { parsePlanTasks } from "../src/plan.ts";
|
|
16
|
+
|
|
17
|
+
const PLAN = `## Tasks
|
|
18
|
+
|
|
19
|
+
- Task-1: parser — files: src/a.ts; wave: 1
|
|
20
|
+
- Task-2: tool — files: src/b.ts; wave: 1
|
|
21
|
+
- Task-3: core — deps: Task-1, Task-2; files: src/c.ts; wave: 2
|
|
22
|
+
- Task-3.1: injection — files: src/c.ts
|
|
23
|
+
|
|
24
|
+
## Verification Checks
|
|
25
|
+
|
|
26
|
+
- [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser ok
|
|
27
|
+
- [ ] \`VC-002\` covers \`Task-2\` and \`Task-3\`; pass condition: tool+core ok
|
|
28
|
+
- [ ] \`VC-003\` covers \`Task-3.1\`; pass condition: injection ok
|
|
29
|
+
`;
|
|
30
|
+
|
|
31
|
+
function checks(done: string[] = []): CheckItem[] {
|
|
32
|
+
const raw = PLAN.split("## Verification Checks")[1] ?? "";
|
|
33
|
+
const items: CheckItem[] = [];
|
|
34
|
+
for (const line of raw.split("\n")) {
|
|
35
|
+
const m = line.match(/^- \[ \] `(VC-\d+)` (.+)$/);
|
|
36
|
+
if (m) items.push({ id: m[1], text: m[2], done: done.includes(m[1]) });
|
|
37
|
+
}
|
|
38
|
+
return items;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
function view(progress: TaskProgressMap = {}) {
|
|
42
|
+
return buildTaskView(parsePlanTasks(PLAN), progress);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
const ALL_DONE: TaskProgressMap = {
|
|
46
|
+
"Task-1": { status: "complete", evidence: "edited src/a.ts" },
|
|
47
|
+
"Task-2": { status: "complete", evidence: "edited src/b.ts" },
|
|
48
|
+
"Task-3": { status: "complete", evidence: "edited src/c.ts" },
|
|
49
|
+
"Task-3.1": { status: "complete", evidence: "injected src/c.ts" },
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
describe("rollback retention", () => {
|
|
53
|
+
it("keeps the evidence of a rolled-back complete task", () => {
|
|
54
|
+
const tasks = view(ALL_DONE);
|
|
55
|
+
const reopened = auditRollbackSet(tasks, checks(), "VC-001");
|
|
56
|
+
assert.deepEqual(reopened, ["Task-1"]);
|
|
57
|
+
const task = tasks[0];
|
|
58
|
+
assert.equal(task.status, "pending");
|
|
59
|
+
// Regression: rollback used to clear evidence, so the agent lost the
|
|
60
|
+
// record of what the previous attempt did and had to re-derive it.
|
|
61
|
+
assert.equal(task.evidence, "edited src/a.ts");
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
it("keeps evidence but clears the skip reason of a rolled-back skipped task", () => {
|
|
65
|
+
const tasks = view({ "Task-2": { status: "skipped", skipReason: "covered by Task-3" } });
|
|
66
|
+
assert.deepEqual(auditRollbackSet(tasks, checks(), "VC-002"), ["Task-2"]);
|
|
67
|
+
const task = tasks.find((t) => t.id === "Task-2");
|
|
68
|
+
assert.equal(task.status, "pending");
|
|
69
|
+
assert.equal(task.skipReason, undefined, "the audit overturned the skip");
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
it("leaves tasks outside the covers clause untouched", () => {
|
|
73
|
+
const tasks = view(ALL_DONE);
|
|
74
|
+
auditRollbackSet(tasks, checks(), "VC-001");
|
|
75
|
+
assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "complete");
|
|
76
|
+
assert.equal(tasks.find((t) => t.id === "Task-2")?.evidence, "edited src/b.ts");
|
|
77
|
+
});
|
|
78
|
+
|
|
79
|
+
it("does not report a task that is already pending as reopened", () => {
|
|
80
|
+
const tasks = view({ "Task-1": { status: "pending" } });
|
|
81
|
+
assert.deepEqual(auditRollbackSet(tasks, checks(), "VC-001"), []);
|
|
82
|
+
});
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
describe("progress persistence", () => {
|
|
86
|
+
it("records every task, including untouched pending ones", () => {
|
|
87
|
+
// Regression: taskProgressMap skipped pending tasks that had no
|
|
88
|
+
// evidence, so after a rollback the checkpoint held zero records and
|
|
89
|
+
// the run's whole history vanished.
|
|
90
|
+
const tasks = view({ "Task-1": { status: "complete", evidence: "done" } });
|
|
91
|
+
const map = taskProgressMap(tasks);
|
|
92
|
+
assert.deepEqual(Object.keys(map).sort(), ["Task-1", "Task-2", "Task-3", "Task-3.1"]);
|
|
93
|
+
assert.equal(map["Task-1"]?.status, "complete");
|
|
94
|
+
assert.equal(map["Task-2"]?.status, "pending");
|
|
95
|
+
assert.equal(map["Task-2"]?.evidence, undefined);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
it("records a rolled-back task as pending with its retained evidence", () => {
|
|
99
|
+
const tasks = view(ALL_DONE);
|
|
100
|
+
auditRollbackSet(tasks, checks(), "VC-001");
|
|
101
|
+
const map = taskProgressMap(tasks);
|
|
102
|
+
assert.equal(map["Task-1"]?.status, "pending");
|
|
103
|
+
assert.equal(map["Task-1"]?.evidence, "edited src/a.ts");
|
|
104
|
+
assert.equal(map["Task-2"]?.status, "complete");
|
|
105
|
+
});
|
|
106
|
+
|
|
107
|
+
it("round-trips through buildTaskView without loss", () => {
|
|
108
|
+
const tasks = view(ALL_DONE);
|
|
109
|
+
auditRollbackSet(tasks, checks(), "VC-001");
|
|
110
|
+
const rebuilt = buildTaskView(parsePlanTasks(PLAN), taskProgressMap(tasks));
|
|
111
|
+
assert.equal(rebuilt[0]?.status, "pending");
|
|
112
|
+
assert.equal(rebuilt[0]?.evidence, "edited src/a.ts");
|
|
113
|
+
});
|
|
114
|
+
});
|
|
115
|
+
|
|
116
|
+
describe("done-flag invalidation", () => {
|
|
117
|
+
it("clears done on checks whose covered tasks were reopened", () => {
|
|
118
|
+
// VC-002 covers Task-2 and Task-3; its rollback reopens those, so a
|
|
119
|
+
// presolved VC-003 (covering Task-3.1, cascaded open too) must lose
|
|
120
|
+
// its satisfied state rather than keep claiming the work is verified.
|
|
121
|
+
const checklist = checks(["VC-001", "VC-003"]);
|
|
122
|
+
const tasks = view(ALL_DONE);
|
|
123
|
+
const reopened = auditRollbackSet(tasks, checklist, "VC-002");
|
|
124
|
+
assert.deepEqual(reopened.sort(), ["Task-2", "Task-3", "Task-3.1"]);
|
|
125
|
+
const cleared = invalidateChecksForRolledBackTasks(checklist, tasks, reopened);
|
|
126
|
+
assert.deepEqual(cleared.sort(), ["VC-003"]);
|
|
127
|
+
assert.equal(checklist.find((c) => c.id === "VC-001")?.done, true, "unaffected check keeps its state");
|
|
128
|
+
});
|
|
129
|
+
|
|
130
|
+
it("clears done on the failing check itself when it is covered by the rollback", () => {
|
|
131
|
+
const checklist = checks(["VC-002"]);
|
|
132
|
+
const tasks = view(ALL_DONE);
|
|
133
|
+
const reopened = auditRollbackSet(tasks, checklist, "VC-002");
|
|
134
|
+
assert.deepEqual(invalidateChecksForRolledBackTasks(checklist, tasks, reopened).sort(), ["VC-002"]);
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
it("leaves every check done when nothing was reopened", () => {
|
|
138
|
+
const checklist = checks(["VC-001", "VC-003"]);
|
|
139
|
+
assert.deepEqual(invalidateChecksForRolledBackTasks(checklist, view(ALL_DONE), []), []);
|
|
140
|
+
assert.deepEqual(checklist.filter((c) => c.done).map((c) => c.id).sort(), ["VC-001", "VC-003"]);
|
|
141
|
+
});
|
|
142
|
+
});
|
|
@@ -418,3 +418,68 @@ describe("state machine reducers", () => {
|
|
|
418
418
|
});
|
|
419
419
|
});
|
|
420
420
|
|
|
421
|
+
|
|
422
|
+
describe("blocking budget persistence", () => {
|
|
423
|
+
function executingCheckpoint(workdir: string, runId: string) {
|
|
424
|
+
createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
|
|
425
|
+
// planIdentityOf hashes the file, so it has to exist.
|
|
426
|
+
const planPath = path.join(workdir, "PLAN_v1.md");
|
|
427
|
+
fs.writeFileSync(planPath, "# PLAN_v1 - demo\n", "utf8");
|
|
428
|
+
const plan = planIdentityOf(planPath, 1);
|
|
429
|
+
let cp = applyPlanWritten(baseCheckpoint(workdir, runId), plan);
|
|
430
|
+
cp = { ...cp, nextAction: "accept-execute" };
|
|
431
|
+
cp = applyExecutionApproved(cp, {
|
|
432
|
+
plan,
|
|
433
|
+
worktree: cp.worktreeRoot,
|
|
434
|
+
headAtApproval: null,
|
|
435
|
+
approvedAt: cp.updatedAt,
|
|
436
|
+
});
|
|
437
|
+
return applyExecutionProgress(cp, { tasks: { "Task-1": { status: "pending" } } });
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
it("round-trips the watchdog budget and the undeterminable set", () => {
|
|
441
|
+
// Without this the budget lived only in memory: a session restart
|
|
442
|
+
// silently handed a stalled run three fresh rounds.
|
|
443
|
+
const { workdir, runId } = setupRun("budget-roundtrip");
|
|
444
|
+
const cp = applyExecutionProgress(executingCheckpoint(workdir, runId), {
|
|
445
|
+
stallRounds: 2,
|
|
446
|
+
audit: { rounds: 1, undeterminable: ["VC-002", "VC-003"] },
|
|
447
|
+
});
|
|
448
|
+
mutateCheckpoint(workdir, runId, () => cp);
|
|
449
|
+
const loaded = loadCheckpoint(workdir, runId);
|
|
450
|
+
assert.ok(loaded.status === "ok");
|
|
451
|
+
assert.equal(loaded.checkpoint.execution?.stallRounds, 2);
|
|
452
|
+
assert.deepEqual(loaded.checkpoint.execution?.audit?.undeterminable, ["VC-002", "VC-003"]);
|
|
453
|
+
assert.equal(loaded.checkpoint.execution?.audit?.rounds, 1);
|
|
454
|
+
});
|
|
455
|
+
|
|
456
|
+
it("loads a checkpoint written before either field existed", () => {
|
|
457
|
+
// Backward compatibility: the loader whitelists keys, so an older
|
|
458
|
+
// checkpoint lacking stallRounds / undeterminable must still load.
|
|
459
|
+
const { workdir, runId } = setupRun("budget-legacy");
|
|
460
|
+
const cp = executingCheckpoint(workdir, runId);
|
|
461
|
+
mutateCheckpoint(workdir, runId, () => cp);
|
|
462
|
+
const filePath = checkpointFilePath(workdir, runId)!;
|
|
463
|
+
const parsed = JSON.parse(fs.readFileSync(filePath, "utf8")) as Record<string, unknown>;
|
|
464
|
+
delete (parsed.execution as Record<string, unknown>).stallRounds;
|
|
465
|
+
delete ((parsed.execution as Record<string, unknown>).audit as Record<string, unknown>).undeterminable;
|
|
466
|
+
fs.writeFileSync(filePath, JSON.stringify(parsed), "utf8");
|
|
467
|
+
const loaded = loadCheckpoint(workdir, runId);
|
|
468
|
+
assert.ok(loaded.status === "ok", "legacy checkpoint still loads");
|
|
469
|
+
assert.equal(loaded.checkpoint.execution?.stallRounds, undefined);
|
|
470
|
+
assert.equal(loaded.checkpoint.execution?.audit?.undeterminable, undefined);
|
|
471
|
+
assert.equal(loaded.checkpoint.execution?.audit?.rounds, 0);
|
|
472
|
+
});
|
|
473
|
+
|
|
474
|
+
it("still rejects an unknown execution key", () => {
|
|
475
|
+
const { workdir, runId } = setupRun("budget-unknown-key");
|
|
476
|
+
const cp = executingCheckpoint(workdir, runId);
|
|
477
|
+
mutateCheckpoint(workdir, runId, () => cp);
|
|
478
|
+
const filePath = checkpointFilePath(workdir, runId)!;
|
|
479
|
+
const parsed = JSON.parse(fs.readFileSync(filePath, "utf8")) as Record<string, unknown>;
|
|
480
|
+
(parsed.execution as Record<string, unknown>).mysteryField = 1;
|
|
481
|
+
fs.writeFileSync(filePath, JSON.stringify(parsed), "utf8");
|
|
482
|
+
const loaded = loadCheckpoint(workdir, runId);
|
|
483
|
+
assert.equal(loaded.status, "corrupt", "the closed schema is still enforced");
|
|
484
|
+
});
|
|
485
|
+
});
|
package/tools/execute-plan.ts
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* `execute_plan` tool — the execution handoff (v0.6.1). On explicit user
|
|
3
3
|
* approval (no Auto-complete) the extension enters task-tree execution mode:
|
|
4
4
|
* progress flows through `plans_update_task`, the dashboard tracks every
|
|
5
|
-
* task, and the
|
|
5
|
+
* task, and the execution reviewer gates the final pass. Legacy I-### plans
|
|
6
6
|
* parse through the fallback with an upgrade notice.
|
|
7
7
|
*/
|
|
8
8
|
|
|
@@ -130,7 +130,7 @@ export async function executeHandoff(
|
|
|
130
130
|
: "";
|
|
131
131
|
approved = await ctx.ui.confirm(
|
|
132
132
|
"Execute this plan now?",
|
|
133
|
-
`${planPath}\n${items.length} verification check(s) over ${planTasks.tasks.length} task(s):\n${preview}${legacyNote}\n\nExecution mode enables write access; task progress is reported with the plans_update_task tool and gated by the
|
|
133
|
+
`${planPath}\n${items.length} verification check(s) over ${planTasks.tasks.length} task(s):\n${preview}${legacyNote}\n\nExecution mode enables write access; task progress is reported with the plans_update_task tool and gated by the execution reviewer.`,
|
|
134
134
|
);
|
|
135
135
|
}
|
|
136
136
|
if (!approved) {
|
|
@@ -148,7 +148,7 @@ export async function executeHandoff(
|
|
|
148
148
|
status: "executing",
|
|
149
149
|
planPath,
|
|
150
150
|
itemCount: items.length,
|
|
151
|
-
message: `${autoNote}Execution approved. ${planTasks.tasks.length} task(s) queued in wave order; report progress with the plans_update_task tool (status + evidence); the
|
|
151
|
+
message: `${autoNote}Execution approved. ${planTasks.tasks.length} task(s) queued in wave order; report progress with the plans_update_task tool (status + evidence); the execution reviewer verifies every check before the run completes.${legacyNote}`,
|
|
152
152
|
};
|
|
153
153
|
}
|
|
154
154
|
|
|
@@ -158,11 +158,18 @@ export async function executeCommand(ctx: ExtensionContext, planPathArg?: string
|
|
|
158
158
|
const planPath = planPathArg ? path.resolve(ctx.cwd, planPathArg.replace(/^@/, "")) : activeExecution?.planPath;
|
|
159
159
|
if (activeExecution && planPath && path.resolve(activeExecution.planPath) === path.resolve(planPath)) {
|
|
160
160
|
const resumed = resumeActiveExecution(ctx);
|
|
161
|
+
// v0.8 phase-aware response: a verifying run continues its review loop
|
|
162
|
+
// (this tool is also the ONLY budget-granting surface at a cap pause).
|
|
163
|
+
const statusText = activeExecution.review?.inFlight
|
|
164
|
+
? "Execution review in progress (status verifying); the reviewer round runs in the overlay."
|
|
165
|
+
: (getExecution()?.stall.paused ?? false)
|
|
166
|
+
? "Execution review paused at the round cap — this confirmation granted a fresh five-round budget; the review resumes now."
|
|
167
|
+
: "Execution resumed; task progress preserved.";
|
|
161
168
|
return {
|
|
162
169
|
status: "executing",
|
|
163
170
|
planPath,
|
|
164
171
|
itemCount: activeExecution.items.length,
|
|
165
|
-
message: resumed ?
|
|
172
|
+
message: resumed ? statusText : "This plan is already executing.",
|
|
166
173
|
};
|
|
167
174
|
}
|
|
168
175
|
return executeHandoff(ctx, planPathArg);
|
|
@@ -173,7 +180,7 @@ export function registerExecutePlanTool(ext: ExtensionAPI): void {
|
|
|
173
180
|
name: "execute_plan",
|
|
174
181
|
label: "Execute Plan",
|
|
175
182
|
description:
|
|
176
|
-
"Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters task-tree execution mode: every task's progress is reported via the plans_update_task tool (status + evidence), the task dashboard tracks the tree (Ctrl+Shift+T expands it), and an independent
|
|
183
|
+
"Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters task-tree execution mode: every task's progress is reported via the plans_update_task tool (status + evidence), the task dashboard tracks the tree (Ctrl+Shift+T expands it), and an independent execution reviewer verifies the verification checks before the run completes. Legacy I-### plans parse through the compatibility mapping with an upgrade notice. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
|
|
177
184
|
promptSnippet: "Hand an accepted plan off to the tracked execution loop",
|
|
178
185
|
parameters: ExecutePlanParams,
|
|
179
186
|
|
package/tools/plans.ts
CHANGED
|
@@ -99,7 +99,7 @@ const PlansParams = Type.Object({
|
|
|
99
99
|
runId: Type.Optional(Type.String()),
|
|
100
100
|
status: Type.Optional(
|
|
101
101
|
StringEnum(
|
|
102
|
-
["planning", "accepted", "executing", "stopped", "abandoned", "done"] as const,
|
|
102
|
+
["planning", "accepted", "executing", "verifying", "stopped", "abandoned", "done"] as const,
|
|
103
103
|
{ description: "set-status: run lifecycle status" },
|
|
104
104
|
),
|
|
105
105
|
),
|