pi-plans 0.7.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,142 @@
1
+ /** Tests for the task tree primitives that the execution review drives:
2
+ * rollback retention, persistence completeness, and done-flag invalidation.
3
+ */
4
+
5
+ import * as assert from "node:assert/strict";
6
+ import { describe, it } from "node:test";
7
+ import type { CheckItem } from "../src/plan.ts";
8
+ import {
9
+ auditRollbackSet,
10
+ buildTaskView,
11
+ invalidateChecksForRolledBackTasks,
12
+ taskProgressMap,
13
+ type TaskProgressMap,
14
+ } from "../src/tasks.ts";
15
+ import { parsePlanTasks } from "../src/plan.ts";
16
+
17
+ const PLAN = `## Tasks
18
+
19
+ - Task-1: parser — files: src/a.ts; wave: 1
20
+ - Task-2: tool — files: src/b.ts; wave: 1
21
+ - Task-3: core — deps: Task-1, Task-2; files: src/c.ts; wave: 2
22
+ - Task-3.1: injection — files: src/c.ts
23
+
24
+ ## Verification Checks
25
+
26
+ - [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser ok
27
+ - [ ] \`VC-002\` covers \`Task-2\` and \`Task-3\`; pass condition: tool+core ok
28
+ - [ ] \`VC-003\` covers \`Task-3.1\`; pass condition: injection ok
29
+ `;
30
+
31
+ function checks(done: string[] = []): CheckItem[] {
32
+ const raw = PLAN.split("## Verification Checks")[1] ?? "";
33
+ const items: CheckItem[] = [];
34
+ for (const line of raw.split("\n")) {
35
+ const m = line.match(/^- \[ \] `(VC-\d+)` (.+)$/);
36
+ if (m) items.push({ id: m[1], text: m[2], done: done.includes(m[1]) });
37
+ }
38
+ return items;
39
+ }
40
+
41
+ function view(progress: TaskProgressMap = {}) {
42
+ return buildTaskView(parsePlanTasks(PLAN), progress);
43
+ }
44
+
45
+ const ALL_DONE: TaskProgressMap = {
46
+ "Task-1": { status: "complete", evidence: "edited src/a.ts" },
47
+ "Task-2": { status: "complete", evidence: "edited src/b.ts" },
48
+ "Task-3": { status: "complete", evidence: "edited src/c.ts" },
49
+ "Task-3.1": { status: "complete", evidence: "injected src/c.ts" },
50
+ };
51
+
52
+ describe("rollback retention", () => {
53
+ it("keeps the evidence of a rolled-back complete task", () => {
54
+ const tasks = view(ALL_DONE);
55
+ const reopened = auditRollbackSet(tasks, checks(), "VC-001");
56
+ assert.deepEqual(reopened, ["Task-1"]);
57
+ const task = tasks[0];
58
+ assert.equal(task.status, "pending");
59
+ // Regression: rollback used to clear evidence, so the agent lost the
60
+ // record of what the previous attempt did and had to re-derive it.
61
+ assert.equal(task.evidence, "edited src/a.ts");
62
+ });
63
+
64
+ it("keeps evidence but clears the skip reason of a rolled-back skipped task", () => {
65
+ const tasks = view({ "Task-2": { status: "skipped", skipReason: "covered by Task-3" } });
66
+ assert.deepEqual(auditRollbackSet(tasks, checks(), "VC-002"), ["Task-2"]);
67
+ const task = tasks.find((t) => t.id === "Task-2");
68
+ assert.equal(task.status, "pending");
69
+ assert.equal(task.skipReason, undefined, "the audit overturned the skip");
70
+ });
71
+
72
+ it("leaves tasks outside the covers clause untouched", () => {
73
+ const tasks = view(ALL_DONE);
74
+ auditRollbackSet(tasks, checks(), "VC-001");
75
+ assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "complete");
76
+ assert.equal(tasks.find((t) => t.id === "Task-2")?.evidence, "edited src/b.ts");
77
+ });
78
+
79
+ it("does not report a task that is already pending as reopened", () => {
80
+ const tasks = view({ "Task-1": { status: "pending" } });
81
+ assert.deepEqual(auditRollbackSet(tasks, checks(), "VC-001"), []);
82
+ });
83
+ });
84
+
85
+ describe("progress persistence", () => {
86
+ it("records every task, including untouched pending ones", () => {
87
+ // Regression: taskProgressMap skipped pending tasks that had no
88
+ // evidence, so after a rollback the checkpoint held zero records and
89
+ // the run's whole history vanished.
90
+ const tasks = view({ "Task-1": { status: "complete", evidence: "done" } });
91
+ const map = taskProgressMap(tasks);
92
+ assert.deepEqual(Object.keys(map).sort(), ["Task-1", "Task-2", "Task-3", "Task-3.1"]);
93
+ assert.equal(map["Task-1"]?.status, "complete");
94
+ assert.equal(map["Task-2"]?.status, "pending");
95
+ assert.equal(map["Task-2"]?.evidence, undefined);
96
+ });
97
+
98
+ it("records a rolled-back task as pending with its retained evidence", () => {
99
+ const tasks = view(ALL_DONE);
100
+ auditRollbackSet(tasks, checks(), "VC-001");
101
+ const map = taskProgressMap(tasks);
102
+ assert.equal(map["Task-1"]?.status, "pending");
103
+ assert.equal(map["Task-1"]?.evidence, "edited src/a.ts");
104
+ assert.equal(map["Task-2"]?.status, "complete");
105
+ });
106
+
107
+ it("round-trips through buildTaskView without loss", () => {
108
+ const tasks = view(ALL_DONE);
109
+ auditRollbackSet(tasks, checks(), "VC-001");
110
+ const rebuilt = buildTaskView(parsePlanTasks(PLAN), taskProgressMap(tasks));
111
+ assert.equal(rebuilt[0]?.status, "pending");
112
+ assert.equal(rebuilt[0]?.evidence, "edited src/a.ts");
113
+ });
114
+ });
115
+
116
+ describe("done-flag invalidation", () => {
117
+ it("clears done on checks whose covered tasks were reopened", () => {
118
+ // VC-002 covers Task-2 and Task-3; its rollback reopens those, so a
119
+ // presolved VC-003 (covering Task-3.1, cascaded open too) must lose
120
+ // its satisfied state rather than keep claiming the work is verified.
121
+ const checklist = checks(["VC-001", "VC-003"]);
122
+ const tasks = view(ALL_DONE);
123
+ const reopened = auditRollbackSet(tasks, checklist, "VC-002");
124
+ assert.deepEqual(reopened.sort(), ["Task-2", "Task-3", "Task-3.1"]);
125
+ const cleared = invalidateChecksForRolledBackTasks(checklist, tasks, reopened);
126
+ assert.deepEqual(cleared.sort(), ["VC-003"]);
127
+ assert.equal(checklist.find((c) => c.id === "VC-001")?.done, true, "unaffected check keeps its state");
128
+ });
129
+
130
+ it("clears done on the failing check itself when it is covered by the rollback", () => {
131
+ const checklist = checks(["VC-002"]);
132
+ const tasks = view(ALL_DONE);
133
+ const reopened = auditRollbackSet(tasks, checklist, "VC-002");
134
+ assert.deepEqual(invalidateChecksForRolledBackTasks(checklist, tasks, reopened).sort(), ["VC-002"]);
135
+ });
136
+
137
+ it("leaves every check done when nothing was reopened", () => {
138
+ const checklist = checks(["VC-001", "VC-003"]);
139
+ assert.deepEqual(invalidateChecksForRolledBackTasks(checklist, view(ALL_DONE), []), []);
140
+ assert.deepEqual(checklist.filter((c) => c.done).map((c) => c.id).sort(), ["VC-001", "VC-003"]);
141
+ });
142
+ });
@@ -418,3 +418,68 @@ describe("state machine reducers", () => {
418
418
  });
419
419
  });
420
420
 
421
+
422
+ describe("blocking budget persistence", () => {
423
+ function executingCheckpoint(workdir: string, runId: string) {
424
+ createCheckpoint(workdir, { runId, originWorkdir: workdir, workdir });
425
+ // planIdentityOf hashes the file, so it has to exist.
426
+ const planPath = path.join(workdir, "PLAN_v1.md");
427
+ fs.writeFileSync(planPath, "# PLAN_v1 - demo\n", "utf8");
428
+ const plan = planIdentityOf(planPath, 1);
429
+ let cp = applyPlanWritten(baseCheckpoint(workdir, runId), plan);
430
+ cp = { ...cp, nextAction: "accept-execute" };
431
+ cp = applyExecutionApproved(cp, {
432
+ plan,
433
+ worktree: cp.worktreeRoot,
434
+ headAtApproval: null,
435
+ approvedAt: cp.updatedAt,
436
+ });
437
+ return applyExecutionProgress(cp, { tasks: { "Task-1": { status: "pending" } } });
438
+ }
439
+
440
+ it("round-trips the watchdog budget and the undeterminable set", () => {
441
+ // Without this the budget lived only in memory: a session restart
442
+ // silently handed a stalled run three fresh rounds.
443
+ const { workdir, runId } = setupRun("budget-roundtrip");
444
+ const cp = applyExecutionProgress(executingCheckpoint(workdir, runId), {
445
+ stallRounds: 2,
446
+ audit: { rounds: 1, undeterminable: ["VC-002", "VC-003"] },
447
+ });
448
+ mutateCheckpoint(workdir, runId, () => cp);
449
+ const loaded = loadCheckpoint(workdir, runId);
450
+ assert.ok(loaded.status === "ok");
451
+ assert.equal(loaded.checkpoint.execution?.stallRounds, 2);
452
+ assert.deepEqual(loaded.checkpoint.execution?.audit?.undeterminable, ["VC-002", "VC-003"]);
453
+ assert.equal(loaded.checkpoint.execution?.audit?.rounds, 1);
454
+ });
455
+
456
+ it("loads a checkpoint written before either field existed", () => {
457
+ // Backward compatibility: the loader whitelists keys, so an older
458
+ // checkpoint lacking stallRounds / undeterminable must still load.
459
+ const { workdir, runId } = setupRun("budget-legacy");
460
+ const cp = executingCheckpoint(workdir, runId);
461
+ mutateCheckpoint(workdir, runId, () => cp);
462
+ const filePath = checkpointFilePath(workdir, runId)!;
463
+ const parsed = JSON.parse(fs.readFileSync(filePath, "utf8")) as Record<string, unknown>;
464
+ delete (parsed.execution as Record<string, unknown>).stallRounds;
465
+ delete ((parsed.execution as Record<string, unknown>).audit as Record<string, unknown>).undeterminable;
466
+ fs.writeFileSync(filePath, JSON.stringify(parsed), "utf8");
467
+ const loaded = loadCheckpoint(workdir, runId);
468
+ assert.ok(loaded.status === "ok", "legacy checkpoint still loads");
469
+ assert.equal(loaded.checkpoint.execution?.stallRounds, undefined);
470
+ assert.equal(loaded.checkpoint.execution?.audit?.undeterminable, undefined);
471
+ assert.equal(loaded.checkpoint.execution?.audit?.rounds, 0);
472
+ });
473
+
474
+ it("still rejects an unknown execution key", () => {
475
+ const { workdir, runId } = setupRun("budget-unknown-key");
476
+ const cp = executingCheckpoint(workdir, runId);
477
+ mutateCheckpoint(workdir, runId, () => cp);
478
+ const filePath = checkpointFilePath(workdir, runId)!;
479
+ const parsed = JSON.parse(fs.readFileSync(filePath, "utf8")) as Record<string, unknown>;
480
+ (parsed.execution as Record<string, unknown>).mysteryField = 1;
481
+ fs.writeFileSync(filePath, JSON.stringify(parsed), "utf8");
482
+ const loaded = loadCheckpoint(workdir, runId);
483
+ assert.equal(loaded.status, "corrupt", "the closed schema is still enforced");
484
+ });
485
+ });
@@ -2,7 +2,7 @@
2
2
  * `execute_plan` tool — the execution handoff (v0.6.1). On explicit user
3
3
  * approval (no Auto-complete) the extension enters task-tree execution mode:
4
4
  * progress flows through `plans_update_task`, the dashboard tracks every
5
- * task, and the completion auditor gates the final pass. Legacy I-### plans
5
+ * task, and the execution reviewer gates the final pass. Legacy I-### plans
6
6
  * parse through the fallback with an upgrade notice.
7
7
  */
8
8
 
@@ -130,7 +130,7 @@ export async function executeHandoff(
130
130
  : "";
131
131
  approved = await ctx.ui.confirm(
132
132
  "Execute this plan now?",
133
- `${planPath}\n${items.length} verification check(s) over ${planTasks.tasks.length} task(s):\n${preview}${legacyNote}\n\nExecution mode enables write access; task progress is reported with the plans_update_task tool and gated by the completion auditor.`,
133
+ `${planPath}\n${items.length} verification check(s) over ${planTasks.tasks.length} task(s):\n${preview}${legacyNote}\n\nExecution mode enables write access; task progress is reported with the plans_update_task tool and gated by the execution reviewer.`,
134
134
  );
135
135
  }
136
136
  if (!approved) {
@@ -148,7 +148,7 @@ export async function executeHandoff(
148
148
  status: "executing",
149
149
  planPath,
150
150
  itemCount: items.length,
151
- message: `${autoNote}Execution approved. ${planTasks.tasks.length} task(s) queued in wave order; report progress with the plans_update_task tool (status + evidence); the completion auditor verifies every check before the run completes.${legacyNote}`,
151
+ message: `${autoNote}Execution approved. ${planTasks.tasks.length} task(s) queued in wave order; report progress with the plans_update_task tool (status + evidence); the execution reviewer verifies every check before the run completes.${legacyNote}`,
152
152
  };
153
153
  }
154
154
 
@@ -158,11 +158,18 @@ export async function executeCommand(ctx: ExtensionContext, planPathArg?: string
158
158
  const planPath = planPathArg ? path.resolve(ctx.cwd, planPathArg.replace(/^@/, "")) : activeExecution?.planPath;
159
159
  if (activeExecution && planPath && path.resolve(activeExecution.planPath) === path.resolve(planPath)) {
160
160
  const resumed = resumeActiveExecution(ctx);
161
+ // v0.8 phase-aware response: a verifying run continues its review loop
162
+ // (this tool is also the ONLY budget-granting surface at a cap pause).
163
+ const statusText = activeExecution.review?.inFlight
164
+ ? "Execution review in progress (status verifying); the reviewer round runs in the overlay."
165
+ : (getExecution()?.stall.paused ?? false)
166
+ ? "Execution review paused at the round cap — this confirmation granted a fresh five-round budget; the review resumes now."
167
+ : "Execution resumed; task progress preserved.";
161
168
  return {
162
169
  status: "executing",
163
170
  planPath,
164
171
  itemCount: activeExecution.items.length,
165
- message: resumed ? "Execution resumed; task progress preserved." : "This plan is already executing.",
172
+ message: resumed ? statusText : "This plan is already executing.",
166
173
  };
167
174
  }
168
175
  return executeHandoff(ctx, planPathArg);
@@ -173,7 +180,7 @@ export function registerExecutePlanTool(ext: ExtensionAPI): void {
173
180
  name: "execute_plan",
174
181
  label: "Execute Plan",
175
182
  description:
176
- "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters task-tree execution mode: every task's progress is reported via the plans_update_task tool (status + evidence), the task dashboard tracks the tree (Ctrl+Shift+T expands it), and an independent completion auditor verifies the verification checks before the run completes. Legacy I-### plans parse through the compatibility mapping with an upgrade notice. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
183
+ "Execution handoff for an accepted plan. Asks the user for explicit approval (never auto-completed), then enters task-tree execution mode: every task's progress is reported via the plans_update_task tool (status + evidence), the task dashboard tracks the tree (Ctrl+Shift+T expands it), and an independent execution reviewer verifies the verification checks before the run completes. Legacy I-### plans parse through the compatibility mapping with an upgrade notice. When several runs with plans exist, a run-picker form selects the target run first. Only call after the user chose 'Execute this plan now' at the handoff question.",
177
184
  promptSnippet: "Hand an accepted plan off to the tracked execution loop",
178
185
  parameters: ExecutePlanParams,
179
186
 
package/tools/plans.ts CHANGED
@@ -99,7 +99,7 @@ const PlansParams = Type.Object({
99
99
  runId: Type.Optional(Type.String()),
100
100
  status: Type.Optional(
101
101
  StringEnum(
102
- ["planning", "accepted", "executing", "stopped", "abandoned", "done"] as const,
102
+ ["planning", "accepted", "executing", "verifying", "stopped", "abandoned", "done"] as const,
103
103
  { description: "set-status: run lifecycle status" },
104
104
  ),
105
105
  ),