pi-plans 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/AGENTS.md +58 -0
  2. package/CONTRIBUTING.md +8 -15
  3. package/README.md +39 -37
  4. package/agents/execution-reviewer.md +40 -0
  5. package/agents/reviewer.md +12 -3
  6. package/index.ts +55 -58
  7. package/package.json +2 -1
  8. package/references/pi-planning-workflow.md +50 -60
  9. package/references/plan-artifact-template.md +81 -60
  10. package/references/state-and-config.md +60 -44
  11. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  12. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  13. package/scripts/run-tests.ts +12 -1
  14. package/scripts/validate.ts +39 -11
  15. package/skills/debug-and-plan/SKILL.md +3 -3
  16. package/skills/plan-big/SKILL.md +3 -3
  17. package/skills/plan-normal/SKILL.md +3 -3
  18. package/skills/plan-small/SKILL.md +4 -4
  19. package/skills/plan-with-refs/SKILL.md +6 -6
  20. package/skills/planning/SKILL.md +1 -1
  21. package/src/ask-form.ts +4 -4
  22. package/src/auditor.ts +227 -0
  23. package/src/auto-approve.ts +1 -1
  24. package/src/autocomplete.ts +19 -17
  25. package/src/code-graph/commands.ts +8 -3
  26. package/src/code-graph/community.ts +1 -1
  27. package/src/code-graph/paths.ts +1 -1
  28. package/src/code-graph/watch.ts +2 -2
  29. package/src/compaction.ts +3 -3
  30. package/src/config-command.ts +146 -73
  31. package/src/dashboard.ts +303 -0
  32. package/src/exec.ts +1185 -924
  33. package/src/global-state.ts +304 -0
  34. package/src/guard.ts +18 -19
  35. package/src/messaging.ts +44 -0
  36. package/src/plan.ts +421 -112
  37. package/src/query-hook.ts +4 -4
  38. package/src/refine-prompts.ts +12 -70
  39. package/src/refine-ui-helpers.ts +24 -5
  40. package/src/refine-ui-state.ts +1 -1
  41. package/src/refine-ui.ts +19 -3
  42. package/src/resume-command.ts +45 -129
  43. package/src/resume.ts +5 -1
  44. package/src/role-panels.ts +542 -0
  45. package/src/run-context.ts +3 -10
  46. package/src/staleness.ts +53 -0
  47. package/src/state.ts +273 -72
  48. package/src/subagent.ts +19 -29
  49. package/src/task-tool.ts +100 -0
  50. package/src/tasks.ts +223 -0
  51. package/src/thinking-levels.ts +67 -0
  52. package/src/ui-language.ts +7 -54
  53. package/src/workflow-state.ts +76 -58
  54. package/tests/analyze-refs.test.ts +35 -18
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +210 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +402 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec-review-loop.test.ts +331 -0
  75. package/tests/exec.test.ts +771 -1706
  76. package/tests/execute-plan.test.ts +44 -19
  77. package/tests/extension-load.test.ts +48 -0
  78. package/tests/global-state.test.ts +371 -0
  79. package/tests/graph-aware-file-tools.test.ts +5 -5
  80. package/tests/guard.test.ts +1 -1
  81. package/tests/multi-run.test.ts +3 -103
  82. package/tests/plan.test.ts +139 -62
  83. package/tests/plans.test.ts +7 -79
  84. package/tests/refine-prompts.test.ts +20 -71
  85. package/tests/refine-resume.test.ts +27 -22
  86. package/tests/refine-ui.test.ts +6 -15
  87. package/tests/resume-lifecycle.test.ts +41 -22
  88. package/tests/resume.test.ts +39 -81
  89. package/tests/role-panels.test.ts +391 -0
  90. package/tests/run-context.test.ts +1 -1
  91. package/tests/run-ownership.test.ts +1 -1
  92. package/tests/stale-ctx.test.ts +218 -0
  93. package/tests/staleness.test.ts +76 -0
  94. package/tests/state.test.ts +155 -32
  95. package/tests/subagent-thinking.test.ts +65 -0
  96. package/tests/subagent-usage.test.ts +1 -1
  97. package/tests/task-tool.test.ts +61 -0
  98. package/tests/tasks.test.ts +142 -0
  99. package/tests/thinking-levels.test.ts +77 -0
  100. package/tests/ui-language.test.ts +2 -17
  101. package/tests/workflow-state.test.ts +73 -90
  102. package/tools/analyze-refs.ts +67 -32
  103. package/tools/ask-choice.ts +7 -53
  104. package/tools/code-graph.ts +2 -2
  105. package/tools/execute-plan.ts +55 -99
  106. package/tools/graph-aware-file-tools.ts +4 -10
  107. package/tools/plans.ts +41 -67
  108. package/tools/refine.ts +101 -164
  109. package/agents/criticizer.md +0 -18
  110. package/agents/executor.md +0 -26
  111. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  112. package/src/panel.ts +0 -473
  113. package/src/termination-prompt.ts +0 -73
  114. package/tests/goal-wait.test.ts +0 -269
  115. package/tests/panel-i-zero.test.ts +0 -420
  116. package/tests/panel.test.ts +0 -355
@@ -1,137 +1,203 @@
1
+ /**
2
+ * Execution lifecycle on the real Pi host (v0.6.1): an execution whose tasks
3
+ * close through the task-status API completes through the audit gate wired
4
+ * by the loaded extension's turn handlers.
5
+ */
6
+
1
7
  import * as assert from "node:assert/strict";
2
8
  import * as fs from "node:fs";
3
9
  import * as os from "node:os";
4
10
  import * as path from "node:path";
5
- import { after, describe, it } from "node:test";
11
+ import { spawnSync } from "node:child_process";
12
+ import { after, before, describe, it } from "node:test";
6
13
  import { InMemoryCredentialStore, createAssistantMessageEventStream } from "@earendil-works/pi-ai";
7
14
  import { createAgentSession, DefaultResourceLoader, initTheme, ModelRuntime, SessionManager, SettingsManager } from "@earendil-works/pi-coding-agent";
8
- import { Type } from "typebox";
9
15
  import piPlansExtension from "../index.ts";
10
- import { GOAL_WAIT_CUSTOM_TYPE, getExecution, startExecution } from "../src/exec.ts";
16
+ import { __setAuditRunnerForTests } from "../src/exec.ts";
17
+ import { getRun, initState, startRun } from "../src/state.ts";
18
+ import { createCheckpoint, loadCheckpoint } from "../src/workflow-state.ts";
19
+ import { setMessagingApi } from "../src/messaging.ts";
11
20
 
12
21
  initTheme("dark", false);
13
- const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-lifecycle-"));
14
- after(() => fs.rmSync(root, { recursive: true, force: true }));
22
+ const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-exec-life-"));
23
+ after(() => {
24
+ fs.rmSync(root, { recursive: true, force: true });
25
+ __setAuditRunnerForTests(null);
26
+ });
15
27
  let serial = 0;
16
28
 
17
- async function exercise(mode: "tui" | "rpc" | "print" | "json", needsWake: boolean, commandResume = false) {
18
- const cwd = path.join(root, String(++serial));
19
- fs.mkdirSync(cwd);
20
- const settingsManager = SettingsManager.inMemory({ compaction: { enabled: false }, retry: { enabled: false } });
21
- const modelRuntime = await ModelRuntime.create({
22
- credentials: new InMemoryCredentialStore(), modelsPath: null,
23
- modelsStorePath: path.join(cwd, "models-store.json"), allowModelNetwork: false, refreshOnCreate: false,
24
- });
25
- const inputs: any[] = [];
26
- let toolCalls = 0;
27
- modelRuntime.registerProvider("local-lifecycle-test", {
28
- baseUrl: "http://unused.invalid", api: "openai-completions", apiKey: "not-a-real-key",
29
- models: [{ id: "fixture", name: "fixture", reasoning: false, input: ["text"],
30
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 100000, maxTokens: 1024 }],
31
- streamSimple: (model: any, context: any) => {
32
- inputs.push(structuredClone({ messages: context.messages, systemPrompt: context.systemPrompt }));
33
- const call = inputs.length;
34
- assert.ok(call <= 5, "unexpected extra model invocation");
35
- const tool = call <= 2;
36
- const text = call === 3
37
- ? needsWake ? "[DONE:VC-001] More verification remains." : "[DONE:VC-001] [DONE:VC-002]"
38
- : call === 4 && needsWake ? "[DONE:VC-002]" : "Review awaits explicit user approval.";
39
- const message: any = {
40
- role: "assistant", api: model.api, provider: model.provider, model: model.id,
41
- content: tool ? [
42
- ...(commandResume && call === 2 ? [{ type: "text", text: "[DONE:VC-001]" }] : []),
43
- { type: "toolCall", id: `call-${call}`, name: "probe", arguments: {} },
44
- ] : [{ type: "text", text: commandResume && call === 3 ? "Interrupted." : text }],
45
- stopReason: tool ? "toolUse" : commandResume && call === 3 ? "aborted" : "stop", timestamp: Date.now(),
46
- usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15,
47
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
48
- };
49
- const stream = createAssistantMessageEventStream();
50
- stream.push({ type: "start", partial: message });
51
- if (message.stopReason === "aborted") stream.push({ type: "error", reason: "aborted", error: message });
52
- else stream.push({ type: "done", reason: message.stopReason, message });
53
- stream.end(message);
54
- return stream;
55
- },
29
+ const PLAN = `# PLAN_v1 - lifecycle
30
+
31
+ ## Tasks
32
+
33
+ - Task-1: parser — files: src/a.ts; wave: 1
34
+ - Task-2: tool — files: src/b.ts; wave: 1
35
+
36
+ ### Execution Waves
37
+
38
+ - wave 1: Task-1, Task-2 — parallel
39
+
40
+ ## Verification Checks
41
+
42
+ - [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser green
43
+ - [ ] \`VC-002\` covers \`Task-2\`; pass condition: tool green
44
+ `;
45
+
46
+ interface Fixture {
47
+ script: Array<{ kind: "text"; text: string }>;
48
+ calls: { messages: unknown; systemPrompt: string }[];
49
+ }
50
+
51
+ function fixtureRuntime(cwd: string, fixture: Fixture) {
52
+ return ModelRuntime.create({
53
+ credentials: new InMemoryCredentialStore(),
54
+ modelsPath: null,
55
+ modelsStorePath: path.join(cwd, "models-store.json"),
56
+ allowModelNetwork: false,
57
+ refreshOnCreate: false,
58
+ }).then((modelRuntime) => {
59
+ modelRuntime.registerProvider("local-exec-life", {
60
+ baseUrl: "http://unused.invalid",
61
+ api: "openai-completions",
62
+ apiKey: "not-a-real-key",
63
+ models: [
64
+ {
65
+ id: "fixture",
66
+ name: "fixture",
67
+ reasoning: false,
68
+ input: ["text"],
69
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
70
+ contextWindow: 100000,
71
+ maxTokens: 1024,
72
+ },
73
+ ],
74
+ streamSimple: (model: unknown, context: { messages: unknown; systemPrompt: string }) => {
75
+ fixture.calls.push(structuredClone({ messages: context.messages, systemPrompt: context.systemPrompt }));
76
+ const step = fixture.script[Math.min(fixture.calls.length, fixture.script.length) - 1] ?? { kind: "text", text: "done" } as const;
77
+ const message = {
78
+ role: "assistant",
79
+ api: "openai-completions",
80
+ provider: "local-exec-life",
81
+ model: "fixture",
82
+ content: [{ type: "text", text: (step as { text: string }).text }],
83
+ stopReason: "stop",
84
+ timestamp: Date.now(),
85
+ usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
86
+ };
87
+ const stream = createAssistantMessageEventStream();
88
+ stream.push({ type: "start", partial: message });
89
+ stream.push({ type: "done", reason: message.stopReason, message });
90
+ stream.end(message);
91
+ return stream;
92
+ },
93
+ });
94
+ return modelRuntime;
56
95
  });
57
- let beforeAgentStarts = 0;
58
- const events: string[] = [];
59
- const errors: string[] = [];
96
+ }
97
+
98
+ async function makeSession(options: { cwd: string; fixture: Fixture; mode: "rpc" }) {
99
+ const { cwd, fixture, mode } = options;
100
+ const settingsManager = SettingsManager.inMemory({ compaction: { enabled: false }, retry: { enabled: false } });
101
+ const modelRuntime = await fixtureRuntime(cwd, fixture);
60
102
  const loader = new DefaultResourceLoader({
61
- cwd, agentDir: path.join(cwd, "agent"), settingsManager,
62
- noExtensions: true, noSkills: true, noPromptTemplates: true, noThemes: true,
63
- agentsFilesOverride: () => ({ agentsFiles: [] }), systemPromptOverride: () => "Deterministic test.",
64
- extensionFactories: [piPlansExtension, pi => {
65
- pi.on("before_agent_start", () => { beforeAgentStarts++; });
66
- pi.on("session_start", async (_event, ctx) => {
67
- await startExecution(pi, ctx, path.join(cwd, "PLAN_v1.md"), [
68
- { id: "VC-001", text: "first", done: false }, { id: "VC-002", text: "second", done: false },
69
- ]);
70
- });
71
- }],
103
+ cwd,
104
+ agentDir: path.join(cwd, "agent"),
105
+ settingsManager,
106
+ noExtensions: true,
107
+ noSkills: true,
108
+ noPromptTemplates: true,
109
+ noThemes: true,
110
+ agentsFilesOverride: () => ({ agentsFiles: [] }),
111
+ systemPromptOverride: () => "Deterministic test.",
112
+ extensionFactories: [piPlansExtension as never],
72
113
  });
73
114
  await loader.reload();
74
115
  assert.deepEqual(loader.getExtensions().errors, []);
75
116
  const { session } = await createAgentSession({
76
- cwd, agentDir: path.join(cwd, "agent"), modelRuntime,
77
- model: modelRuntime.getModel("local-lifecycle-test", "fixture")!, thinkingLevel: "off",
78
- resourceLoader: loader, settingsManager, sessionManager: SessionManager.inMemory(cwd), tools: ["probe"],
79
- customTools: [{ name: "probe", label: "Probe", description: "Local test probe", parameters: Type.Object({}),
80
- execute: async () => { toolCalls++; return { content: [{ type: "text", text: "ok" }], details: {} }; } }],
117
+ cwd,
118
+ agentDir: path.join(cwd, "agent"),
119
+ modelRuntime,
120
+ model: modelRuntime.getModel("local-exec-life", "fixture")!,
121
+ thinkingLevel: "off",
122
+ resourceLoader: loader,
123
+ settingsManager,
124
+ sessionManager: SessionManager.inMemory(cwd),
125
+ tools: [],
126
+ customTools: [],
81
127
  });
82
- const unsubscribe = session.subscribe(event => events.push(event.type));
83
- try {
84
- await session.bindExtensions({
85
- mode,
86
- ...(mode === "tui" || mode === "rpc" ? { uiContext: {
87
- setStatus: () => {}, notify: () => {}, theme: { fg: (_c: string, s: string) => s },
88
- } as any } : {}),
89
- onError: error => errors.push(error.error),
90
- });
91
- await session.prompt("Implement the test plan.");
92
- await session.waitForIdle();
93
- if (commandResume) {
94
- assert.equal(getExecution()?.goalWait?.paused, true);
95
- assert.deepEqual(getExecution()?.items.map(item => item.done), [true, false]);
96
- assert.equal(inputs.length, 3);
97
- await session.prompt("/plans-execute");
98
- }
99
- // SDK callers, unlike print mode, own the runtime until all nested wakes settle.
100
- await session.waitForIdle();
101
- assert.deepEqual(errors, []);
102
- assert.deepEqual(session.messages.filter((m: any) => m.role === "assistant" && m.stopReason === "error"), [], "fixture model must run successfully");
103
- const wakes = session.messages.filter((m: any) => m.customType === GOAL_WAIT_CUSTOM_TYPE) as any[];
104
- assert.equal(toolCalls, 2);
105
- assert.equal(beforeAgentStarts, 1, "custom wake must work without before_agent_start");
106
- const interactive = mode === "tui" || mode === "rpc";
107
- assert.equal(wakes.length, interactive && needsWake ? 1 : 0);
108
- assert.equal(inputs.length, interactive ? needsWake ? 5 : 4 : 3);
109
- if (interactive && needsWake) {
110
- assert.equal(wakes[0].display, false);
111
- assert.match(JSON.stringify(inputs[3].messages), /1\/2 verifier items done/);
112
- assert.match(wakes[0].content, /- `VC-002` second/);
113
- assert.doesNotMatch(wakes[0].content, /- `VC-001` first/);
114
- }
115
- if (interactive || !needsWake) assert.equal(getExecution(), null);
116
- else assert.deepEqual(getExecution()?.items.map(item => item.done), [true, false]);
117
- assert.equal(session.pendingMessageCount, 0);
118
- assert.equal(events.at(-1), "agent_settled");
119
- return { events, calls: inputs.length, wakes: wakes.length };
120
- } finally {
121
- unsubscribe();
122
- session.dispose();
123
- }
128
+ const errors: string[] = [];
129
+ await session.bindExtensions({
130
+ mode,
131
+ uiContext: {
132
+ setStatus: () => {},
133
+ notify: () => {},
134
+ confirm: async () => true,
135
+ select: async (_t: string, options: string[]) => options[0],
136
+ input: async () => "typed",
137
+ theme: { fg: (_c: string, s: string) => s },
138
+ },
139
+ onError: (error: { error: string }) => errors.push(error.error),
140
+ });
141
+ (session as unknown as { errors: string[] }).errors = errors;
142
+ return session as never as Awaited<ReturnType<typeof import("../index.ts").default>> extends never ? never : typeof session;
124
143
  }
125
144
 
126
- describe("goal-wait on the real Pi host", () => {
127
- it("dispatches the registered /plans-execute command and preserves completed VCs", { timeout: 15000 }, async () => {
128
- await exercise("rpc", true, true);
129
- });
130
- for (const mode of ["tui", "rpc", "print", "json"] as const) {
131
- for (const needsWake of [false, true]) {
132
- it(`${mode}: tools then ${needsWake ? "incomplete stop" : "completion"}`, { timeout: 15000 }, async () => {
133
- await exercise(mode, needsWake);
134
- });
145
+ before(() => {
146
+ // deterministic audit: every check passes
147
+ __setAuditRunnerForTests(async ({ checklist }) => ({
148
+ round: 1,
149
+ passed: checklist.map((item) => {
150
+ item.done = true;
151
+ return item.id;
152
+ }),
153
+ failed: [],
154
+ rolledBack: [],
155
+ report: "all checks pass",
156
+ }));
157
+ });
158
+
159
+ describe("execution lifecycle on the real host", () => {
160
+ it("completes a task-tree execution through the audit gate", { timeout: 60000 }, async () => {
161
+ serial += 1;
162
+ const cwd = path.join(root, String(serial));
163
+ fs.mkdirSync(cwd);
164
+ spawnSync("git", ["init"], { cwd });
165
+ spawnSync("git", ["config", "user.email", "t@e.com"], { cwd });
166
+ spawnSync("git", ["config", "user.name", "T"], { cwd });
167
+ initState(cwd);
168
+ const { run } = startRun(cwd, { topic: "lifecycle", skill: "plan-small", requestText: "x" });
169
+ createCheckpoint(cwd, { runId: run.run_id, originWorkdir: cwd, workdir: cwd });
170
+ fs.mkdirSync(run.artifact_dir, { recursive: true });
171
+ const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
172
+ fs.writeFileSync(planPath, PLAN, "utf8");
173
+ spawnSync("git", ["add", "-A"], { cwd });
174
+ spawnSync("git", ["commit", "-m", "seed"], { cwd });
175
+
176
+ const fixture: Fixture = { script: [{ kind: "text", text: "All tasks verified." }], calls: [] };
177
+ const session = (await makeSession({ cwd, fixture, mode: "rpc" })) as unknown as { prompt: (t: string) => Promise<unknown>; waitForIdle: () => Promise<void>; dispose: () => void; sessionManager: unknown; errors: string[] };
178
+ try {
179
+ // Same-process module instances as the session's extension wiring.
180
+ const { startExecution, getExecution, persistTaskProgress } = await import("../src/exec.ts");
181
+ const { parseChecklist, parsePlanTasks } = await import("../src/plan.ts");
182
+ const { applyTaskUpdate } = await import("../src/task-tool.ts");
183
+ const ctxLike = { cwd, sessionManager: session.sessionManager, hasUI: false, mode: "json", ui: { notify: () => {}, setStatus: () => {}, theme: { fg: (_c: string, t: string) => t } } } as never;
184
+ setMessagingApi({ appendEntry: () => {}, sendMessage: () => {}, sendUserMessage: async () => {} });
185
+ await startExecution(ctxLike, { planPath, planTasks: parsePlanTasks(PLAN), items: parseChecklist(PLAN) });
186
+ applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "parser green");
187
+ applyTaskUpdate(getExecution()!.tasks, "Task-2", "complete", "tool green");
188
+ persistTaskProgress(ctxLike);
189
+
190
+ await session.prompt("Finish the run.");
191
+ await session.waitForIdle();
192
+
193
+ const final = loadCheckpoint(cwd, run.run_id);
194
+ assert.equal(final.status, "ok");
195
+ assert.equal(final.checkpoint.phase, "completed", "audit passed → terminal phase");
196
+ assert.equal(final.checkpoint.execution?.audit?.passed, true);
197
+ assert.deepEqual((final.checkpoint.execution?.doneVcIds ?? []).sort(), ["VC-001", "VC-002"]);
198
+ assert.equal(getRun(cwd, run.run_id)?.status, "done");
199
+ } finally {
200
+ session.dispose();
135
201
  }
136
- }
202
+ });
137
203
  });