pi-plans 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +58 -0
- package/CONTRIBUTING.md +8 -15
- package/README.md +39 -37
- package/agents/execution-reviewer.md +40 -0
- package/agents/reviewer.md +12 -3
- package/index.ts +55 -58
- package/package.json +2 -1
- package/references/pi-planning-workflow.md +50 -60
- package/references/plan-artifact-template.md +81 -60
- package/references/state-and-config.md +60 -44
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +39 -11
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +227 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +8 -3
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +303 -0
- package/src/exec.ts +1185 -924
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +19 -3
- package/src/resume-command.ts +45 -129
- package/src/resume.ts +5 -1
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/staleness.ts +53 -0
- package/src/state.ts +273 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +223 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +7 -54
- package/src/workflow-state.ts +76 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +210 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +402 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec-review-loop.test.ts +331 -0
- package/tests/exec.test.ts +771 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +41 -22
- package/tests/resume.test.ts +39 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/staleness.test.ts +76 -0
- package/tests/state.test.ts +155 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/tasks.test.ts +142 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +73 -90
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +55 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +41 -67
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
|
@@ -1,137 +1,203 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Execution lifecycle on the real Pi host (v0.6.1): an execution whose tasks
|
|
3
|
+
* close through the task-status API completes through the audit gate wired
|
|
4
|
+
* by the loaded extension's turn handlers.
|
|
5
|
+
*/
|
|
6
|
+
|
|
1
7
|
import * as assert from "node:assert/strict";
|
|
2
8
|
import * as fs from "node:fs";
|
|
3
9
|
import * as os from "node:os";
|
|
4
10
|
import * as path from "node:path";
|
|
5
|
-
import {
|
|
11
|
+
import { spawnSync } from "node:child_process";
|
|
12
|
+
import { after, before, describe, it } from "node:test";
|
|
6
13
|
import { InMemoryCredentialStore, createAssistantMessageEventStream } from "@earendil-works/pi-ai";
|
|
7
14
|
import { createAgentSession, DefaultResourceLoader, initTheme, ModelRuntime, SessionManager, SettingsManager } from "@earendil-works/pi-coding-agent";
|
|
8
|
-
import { Type } from "typebox";
|
|
9
15
|
import piPlansExtension from "../index.ts";
|
|
10
|
-
import {
|
|
16
|
+
import { __setAuditRunnerForTests } from "../src/exec.ts";
|
|
17
|
+
import { getRun, initState, startRun } from "../src/state.ts";
|
|
18
|
+
import { createCheckpoint, loadCheckpoint } from "../src/workflow-state.ts";
|
|
19
|
+
import { setMessagingApi } from "../src/messaging.ts";
|
|
11
20
|
|
|
12
21
|
initTheme("dark", false);
|
|
13
|
-
const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-
|
|
14
|
-
after(() =>
|
|
22
|
+
const root = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-exec-life-"));
|
|
23
|
+
after(() => {
|
|
24
|
+
fs.rmSync(root, { recursive: true, force: true });
|
|
25
|
+
__setAuditRunnerForTests(null);
|
|
26
|
+
});
|
|
15
27
|
let serial = 0;
|
|
16
28
|
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
29
|
+
const PLAN = `# PLAN_v1 - lifecycle
|
|
30
|
+
|
|
31
|
+
## Tasks
|
|
32
|
+
|
|
33
|
+
- Task-1: parser — files: src/a.ts; wave: 1
|
|
34
|
+
- Task-2: tool — files: src/b.ts; wave: 1
|
|
35
|
+
|
|
36
|
+
### Execution Waves
|
|
37
|
+
|
|
38
|
+
- wave 1: Task-1, Task-2 — parallel
|
|
39
|
+
|
|
40
|
+
## Verification Checks
|
|
41
|
+
|
|
42
|
+
- [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser green
|
|
43
|
+
- [ ] \`VC-002\` covers \`Task-2\`; pass condition: tool green
|
|
44
|
+
`;
|
|
45
|
+
|
|
46
|
+
interface Fixture {
|
|
47
|
+
script: Array<{ kind: "text"; text: string }>;
|
|
48
|
+
calls: { messages: unknown; systemPrompt: string }[];
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
function fixtureRuntime(cwd: string, fixture: Fixture) {
|
|
52
|
+
return ModelRuntime.create({
|
|
53
|
+
credentials: new InMemoryCredentialStore(),
|
|
54
|
+
modelsPath: null,
|
|
55
|
+
modelsStorePath: path.join(cwd, "models-store.json"),
|
|
56
|
+
allowModelNetwork: false,
|
|
57
|
+
refreshOnCreate: false,
|
|
58
|
+
}).then((modelRuntime) => {
|
|
59
|
+
modelRuntime.registerProvider("local-exec-life", {
|
|
60
|
+
baseUrl: "http://unused.invalid",
|
|
61
|
+
api: "openai-completions",
|
|
62
|
+
apiKey: "not-a-real-key",
|
|
63
|
+
models: [
|
|
64
|
+
{
|
|
65
|
+
id: "fixture",
|
|
66
|
+
name: "fixture",
|
|
67
|
+
reasoning: false,
|
|
68
|
+
input: ["text"],
|
|
69
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
70
|
+
contextWindow: 100000,
|
|
71
|
+
maxTokens: 1024,
|
|
72
|
+
},
|
|
73
|
+
],
|
|
74
|
+
streamSimple: (model: unknown, context: { messages: unknown; systemPrompt: string }) => {
|
|
75
|
+
fixture.calls.push(structuredClone({ messages: context.messages, systemPrompt: context.systemPrompt }));
|
|
76
|
+
const step = fixture.script[Math.min(fixture.calls.length, fixture.script.length) - 1] ?? { kind: "text", text: "done" } as const;
|
|
77
|
+
const message = {
|
|
78
|
+
role: "assistant",
|
|
79
|
+
api: "openai-completions",
|
|
80
|
+
provider: "local-exec-life",
|
|
81
|
+
model: "fixture",
|
|
82
|
+
content: [{ type: "text", text: (step as { text: string }).text }],
|
|
83
|
+
stopReason: "stop",
|
|
84
|
+
timestamp: Date.now(),
|
|
85
|
+
usage: { input: 10, output: 5, cacheRead: 0, cacheWrite: 0, totalTokens: 15, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 } },
|
|
86
|
+
};
|
|
87
|
+
const stream = createAssistantMessageEventStream();
|
|
88
|
+
stream.push({ type: "start", partial: message });
|
|
89
|
+
stream.push({ type: "done", reason: message.stopReason, message });
|
|
90
|
+
stream.end(message);
|
|
91
|
+
return stream;
|
|
92
|
+
},
|
|
93
|
+
});
|
|
94
|
+
return modelRuntime;
|
|
56
95
|
});
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
async function makeSession(options: { cwd: string; fixture: Fixture; mode: "rpc" }) {
|
|
99
|
+
const { cwd, fixture, mode } = options;
|
|
100
|
+
const settingsManager = SettingsManager.inMemory({ compaction: { enabled: false }, retry: { enabled: false } });
|
|
101
|
+
const modelRuntime = await fixtureRuntime(cwd, fixture);
|
|
60
102
|
const loader = new DefaultResourceLoader({
|
|
61
|
-
cwd,
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
}],
|
|
103
|
+
cwd,
|
|
104
|
+
agentDir: path.join(cwd, "agent"),
|
|
105
|
+
settingsManager,
|
|
106
|
+
noExtensions: true,
|
|
107
|
+
noSkills: true,
|
|
108
|
+
noPromptTemplates: true,
|
|
109
|
+
noThemes: true,
|
|
110
|
+
agentsFilesOverride: () => ({ agentsFiles: [] }),
|
|
111
|
+
systemPromptOverride: () => "Deterministic test.",
|
|
112
|
+
extensionFactories: [piPlansExtension as never],
|
|
72
113
|
});
|
|
73
114
|
await loader.reload();
|
|
74
115
|
assert.deepEqual(loader.getExtensions().errors, []);
|
|
75
116
|
const { session } = await createAgentSession({
|
|
76
|
-
cwd,
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
117
|
+
cwd,
|
|
118
|
+
agentDir: path.join(cwd, "agent"),
|
|
119
|
+
modelRuntime,
|
|
120
|
+
model: modelRuntime.getModel("local-exec-life", "fixture")!,
|
|
121
|
+
thinkingLevel: "off",
|
|
122
|
+
resourceLoader: loader,
|
|
123
|
+
settingsManager,
|
|
124
|
+
sessionManager: SessionManager.inMemory(cwd),
|
|
125
|
+
tools: [],
|
|
126
|
+
customTools: [],
|
|
81
127
|
});
|
|
82
|
-
const
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
await session.prompt("/plans-execute");
|
|
98
|
-
}
|
|
99
|
-
// SDK callers, unlike print mode, own the runtime until all nested wakes settle.
|
|
100
|
-
await session.waitForIdle();
|
|
101
|
-
assert.deepEqual(errors, []);
|
|
102
|
-
assert.deepEqual(session.messages.filter((m: any) => m.role === "assistant" && m.stopReason === "error"), [], "fixture model must run successfully");
|
|
103
|
-
const wakes = session.messages.filter((m: any) => m.customType === GOAL_WAIT_CUSTOM_TYPE) as any[];
|
|
104
|
-
assert.equal(toolCalls, 2);
|
|
105
|
-
assert.equal(beforeAgentStarts, 1, "custom wake must work without before_agent_start");
|
|
106
|
-
const interactive = mode === "tui" || mode === "rpc";
|
|
107
|
-
assert.equal(wakes.length, interactive && needsWake ? 1 : 0);
|
|
108
|
-
assert.equal(inputs.length, interactive ? needsWake ? 5 : 4 : 3);
|
|
109
|
-
if (interactive && needsWake) {
|
|
110
|
-
assert.equal(wakes[0].display, false);
|
|
111
|
-
assert.match(JSON.stringify(inputs[3].messages), /1\/2 verifier items done/);
|
|
112
|
-
assert.match(wakes[0].content, /- `VC-002` second/);
|
|
113
|
-
assert.doesNotMatch(wakes[0].content, /- `VC-001` first/);
|
|
114
|
-
}
|
|
115
|
-
if (interactive || !needsWake) assert.equal(getExecution(), null);
|
|
116
|
-
else assert.deepEqual(getExecution()?.items.map(item => item.done), [true, false]);
|
|
117
|
-
assert.equal(session.pendingMessageCount, 0);
|
|
118
|
-
assert.equal(events.at(-1), "agent_settled");
|
|
119
|
-
return { events, calls: inputs.length, wakes: wakes.length };
|
|
120
|
-
} finally {
|
|
121
|
-
unsubscribe();
|
|
122
|
-
session.dispose();
|
|
123
|
-
}
|
|
128
|
+
const errors: string[] = [];
|
|
129
|
+
await session.bindExtensions({
|
|
130
|
+
mode,
|
|
131
|
+
uiContext: {
|
|
132
|
+
setStatus: () => {},
|
|
133
|
+
notify: () => {},
|
|
134
|
+
confirm: async () => true,
|
|
135
|
+
select: async (_t: string, options: string[]) => options[0],
|
|
136
|
+
input: async () => "typed",
|
|
137
|
+
theme: { fg: (_c: string, s: string) => s },
|
|
138
|
+
},
|
|
139
|
+
onError: (error: { error: string }) => errors.push(error.error),
|
|
140
|
+
});
|
|
141
|
+
(session as unknown as { errors: string[] }).errors = errors;
|
|
142
|
+
return session as never as Awaited<ReturnType<typeof import("../index.ts").default>> extends never ? never : typeof session;
|
|
124
143
|
}
|
|
125
144
|
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
145
|
+
before(() => {
|
|
146
|
+
// deterministic audit: every check passes
|
|
147
|
+
__setAuditRunnerForTests(async ({ checklist }) => ({
|
|
148
|
+
round: 1,
|
|
149
|
+
passed: checklist.map((item) => {
|
|
150
|
+
item.done = true;
|
|
151
|
+
return item.id;
|
|
152
|
+
}),
|
|
153
|
+
failed: [],
|
|
154
|
+
rolledBack: [],
|
|
155
|
+
report: "all checks pass",
|
|
156
|
+
}));
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
describe("execution lifecycle on the real host", () => {
|
|
160
|
+
it("completes a task-tree execution through the audit gate", { timeout: 60000 }, async () => {
|
|
161
|
+
serial += 1;
|
|
162
|
+
const cwd = path.join(root, String(serial));
|
|
163
|
+
fs.mkdirSync(cwd);
|
|
164
|
+
spawnSync("git", ["init"], { cwd });
|
|
165
|
+
spawnSync("git", ["config", "user.email", "t@e.com"], { cwd });
|
|
166
|
+
spawnSync("git", ["config", "user.name", "T"], { cwd });
|
|
167
|
+
initState(cwd);
|
|
168
|
+
const { run } = startRun(cwd, { topic: "lifecycle", skill: "plan-small", requestText: "x" });
|
|
169
|
+
createCheckpoint(cwd, { runId: run.run_id, originWorkdir: cwd, workdir: cwd });
|
|
170
|
+
fs.mkdirSync(run.artifact_dir, { recursive: true });
|
|
171
|
+
const planPath = path.join(run.artifact_dir, "PLAN_v1.md");
|
|
172
|
+
fs.writeFileSync(planPath, PLAN, "utf8");
|
|
173
|
+
spawnSync("git", ["add", "-A"], { cwd });
|
|
174
|
+
spawnSync("git", ["commit", "-m", "seed"], { cwd });
|
|
175
|
+
|
|
176
|
+
const fixture: Fixture = { script: [{ kind: "text", text: "All tasks verified." }], calls: [] };
|
|
177
|
+
const session = (await makeSession({ cwd, fixture, mode: "rpc" })) as unknown as { prompt: (t: string) => Promise<unknown>; waitForIdle: () => Promise<void>; dispose: () => void; sessionManager: unknown; errors: string[] };
|
|
178
|
+
try {
|
|
179
|
+
// Same-process module instances as the session's extension wiring.
|
|
180
|
+
const { startExecution, getExecution, persistTaskProgress } = await import("../src/exec.ts");
|
|
181
|
+
const { parseChecklist, parsePlanTasks } = await import("../src/plan.ts");
|
|
182
|
+
const { applyTaskUpdate } = await import("../src/task-tool.ts");
|
|
183
|
+
const ctxLike = { cwd, sessionManager: session.sessionManager, hasUI: false, mode: "json", ui: { notify: () => {}, setStatus: () => {}, theme: { fg: (_c: string, t: string) => t } } } as never;
|
|
184
|
+
setMessagingApi({ appendEntry: () => {}, sendMessage: () => {}, sendUserMessage: async () => {} });
|
|
185
|
+
await startExecution(ctxLike, { planPath, planTasks: parsePlanTasks(PLAN), items: parseChecklist(PLAN) });
|
|
186
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-1", "complete", "parser green");
|
|
187
|
+
applyTaskUpdate(getExecution()!.tasks, "Task-2", "complete", "tool green");
|
|
188
|
+
persistTaskProgress(ctxLike);
|
|
189
|
+
|
|
190
|
+
await session.prompt("Finish the run.");
|
|
191
|
+
await session.waitForIdle();
|
|
192
|
+
|
|
193
|
+
const final = loadCheckpoint(cwd, run.run_id);
|
|
194
|
+
assert.equal(final.status, "ok");
|
|
195
|
+
assert.equal(final.checkpoint.phase, "completed", "audit passed → terminal phase");
|
|
196
|
+
assert.equal(final.checkpoint.execution?.audit?.passed, true);
|
|
197
|
+
assert.deepEqual((final.checkpoint.execution?.doneVcIds ?? []).sort(), ["VC-001", "VC-002"]);
|
|
198
|
+
assert.equal(getRun(cwd, run.run_id)?.status, "done");
|
|
199
|
+
} finally {
|
|
200
|
+
session.dispose();
|
|
135
201
|
}
|
|
136
|
-
}
|
|
202
|
+
});
|
|
137
203
|
});
|