pi-plans 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +3 -3
- package/README.md +39 -37
- package/agents/reviewer.md +12 -3
- package/index.ts +42 -35
- package/package.json +1 -1
- package/references/pi-planning-workflow.md +44 -60
- package/references/plan-artifact-template.md +71 -60
- package/references/state-and-config.md +59 -43
- package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
- package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
- package/scripts/run-tests.ts +12 -1
- package/scripts/validate.ts +20 -9
- package/skills/debug-and-plan/SKILL.md +3 -3
- package/skills/plan-big/SKILL.md +3 -3
- package/skills/plan-normal/SKILL.md +3 -3
- package/skills/plan-small/SKILL.md +4 -4
- package/skills/plan-with-refs/SKILL.md +6 -6
- package/skills/planning/SKILL.md +1 -1
- package/src/ask-form.ts +4 -4
- package/src/auditor.ts +126 -0
- package/src/auto-approve.ts +1 -1
- package/src/autocomplete.ts +19 -17
- package/src/code-graph/commands.ts +2 -2
- package/src/code-graph/community.ts +1 -1
- package/src/code-graph/paths.ts +1 -1
- package/src/code-graph/watch.ts +2 -2
- package/src/compaction.ts +3 -3
- package/src/config-command.ts +146 -73
- package/src/dashboard.ts +257 -0
- package/src/exec.ts +692 -919
- package/src/global-state.ts +304 -0
- package/src/guard.ts +18 -19
- package/src/messaging.ts +44 -0
- package/src/plan.ts +421 -112
- package/src/query-hook.ts +4 -4
- package/src/refine-prompts.ts +12 -70
- package/src/refine-ui-helpers.ts +24 -5
- package/src/refine-ui-state.ts +1 -1
- package/src/refine-ui.ts +1 -1
- package/src/resume-command.ts +34 -128
- package/src/role-panels.ts +542 -0
- package/src/run-context.ts +3 -10
- package/src/state.ts +272 -72
- package/src/subagent.ts +19 -29
- package/src/task-tool.ts +100 -0
- package/src/tasks.ts +189 -0
- package/src/thinking-levels.ts +67 -0
- package/src/ui-language.ts +3 -54
- package/src/workflow-state.ts +63 -58
- package/tests/analyze-refs.test.ts +35 -18
- package/tests/ask-choice-schema.test.ts +0 -12
- package/tests/ask-choice.test.ts +2 -49
- package/tests/ask-form-tool.test.ts +4 -5
- package/tests/ask-form.test.ts +2 -2
- package/tests/auditor.test.ts +111 -0
- package/tests/auto-approve.test.ts +7 -10
- package/tests/autocomplete.test.ts +8 -11
- package/tests/code-graph-apply-action.test.ts +2 -2
- package/tests/code-graph-commands.test.ts +2 -2
- package/tests/code-graph-index.test.ts +2 -2
- package/tests/code-graph-loop.e2e.test.ts +1 -1
- package/tests/code-graph-mutations.test.ts +1 -1
- package/tests/code-graph-rollback.test.ts +1 -1
- package/tests/code-graph-v05.test.ts +2 -2
- package/tests/compaction.test.ts +1 -1
- package/tests/config-command.test.ts +103 -100
- package/tests/dashboard.test.ts +268 -0
- package/tests/exec-lifecycle.test.ts +181 -115
- package/tests/exec-panel-lifecycle.test.ts +106 -251
- package/tests/exec.test.ts +617 -1706
- package/tests/execute-plan.test.ts +44 -19
- package/tests/extension-load.test.ts +48 -0
- package/tests/global-state.test.ts +371 -0
- package/tests/graph-aware-file-tools.test.ts +5 -5
- package/tests/guard.test.ts +1 -1
- package/tests/multi-run.test.ts +3 -103
- package/tests/plan.test.ts +139 -62
- package/tests/plans.test.ts +7 -79
- package/tests/refine-prompts.test.ts +20 -71
- package/tests/refine-resume.test.ts +27 -22
- package/tests/refine-ui.test.ts +6 -15
- package/tests/resume-lifecycle.test.ts +37 -22
- package/tests/resume.test.ts +33 -81
- package/tests/role-panels.test.ts +391 -0
- package/tests/run-context.test.ts +1 -1
- package/tests/run-ownership.test.ts +1 -1
- package/tests/stale-ctx.test.ts +218 -0
- package/tests/state.test.ts +151 -32
- package/tests/subagent-thinking.test.ts +65 -0
- package/tests/subagent-usage.test.ts +1 -1
- package/tests/task-tool.test.ts +61 -0
- package/tests/thinking-levels.test.ts +77 -0
- package/tests/ui-language.test.ts +2 -17
- package/tests/workflow-state.test.ts +17 -99
- package/tools/analyze-refs.ts +67 -32
- package/tools/ask-choice.ts +7 -53
- package/tools/code-graph.ts +2 -2
- package/tools/execute-plan.ts +48 -99
- package/tools/graph-aware-file-tools.ts +4 -10
- package/tools/plans.ts +40 -66
- package/tools/refine.ts +101 -164
- package/agents/criticizer.md +0 -18
- package/agents/executor.md +0 -26
- package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
- package/src/panel.ts +0 -473
- package/src/termination-prompt.ts +0 -73
- package/tests/goal-wait.test.ts +0 -269
- package/tests/panel-i-zero.test.ts +0 -420
- package/tests/panel.test.ts +0 -355
package/tests/multi-run.test.ts
CHANGED
|
@@ -19,9 +19,7 @@ import { planningWriteBlockReason } from "../src/guard.ts";
|
|
|
19
19
|
import { executionCandidates, abandonCandidates, runPickerLabel } from "../src/run-picker.ts";
|
|
20
20
|
import { subagentChildEnv } from "../src/subagent.ts";
|
|
21
21
|
import {
|
|
22
|
-
applyDoneMarkers,
|
|
23
22
|
getExecution,
|
|
24
|
-
mirrorDelegateMarkers,
|
|
25
23
|
startExecution,
|
|
26
24
|
stopExecution,
|
|
27
25
|
} from "../src/exec.ts";
|
|
@@ -68,7 +66,7 @@ describe("run registry (v0.6.0)", () => {
|
|
|
68
66
|
const first = startRun(workdir, { topic: "first", skill: "plan-small", requestText: "a" }).run;
|
|
69
67
|
const second = startRun(workdir, { topic: "second", skill: "plan-big", requestText: "b" }).run;
|
|
70
68
|
// Corrupt a third run's run.json: the scan must skip it, never throw.
|
|
71
|
-
const stateRoot = path.join(workdir, ".git", "
|
|
69
|
+
const stateRoot = path.join(workdir, ".git", "pi-plans");
|
|
72
70
|
fs.mkdirSync(path.join(stateRoot, "runs", "corrupt-run-id"), { recursive: true });
|
|
73
71
|
fs.writeFileSync(path.join(stateRoot, "runs", "corrupt-run-id", "run.json"), "{ broken", "utf8");
|
|
74
72
|
|
|
@@ -85,7 +83,7 @@ describe("run registry (v0.6.0)", () => {
|
|
|
85
83
|
const workdir = freshWorkdir();
|
|
86
84
|
initState(workdir);
|
|
87
85
|
startRun(workdir, { topic: "pointerless", skill: "plan-small", requestText: "x" });
|
|
88
|
-
const activePath = path.join(workdir, ".git", "
|
|
86
|
+
const activePath = path.join(workdir, ".git", "pi-plans", "active.json");
|
|
89
87
|
assert.equal(fs.existsSync(activePath), false, "registry workdirs keep no shared pointer");
|
|
90
88
|
});
|
|
91
89
|
|
|
@@ -103,7 +101,7 @@ describe("run registry (v0.6.0)", () => {
|
|
|
103
101
|
it("readActive falls back to a legacy active.json only when the scan finds nothing", () => {
|
|
104
102
|
const workdir = freshWorkdir();
|
|
105
103
|
initState(workdir);
|
|
106
|
-
const stateRoot = path.join(workdir, ".git", "
|
|
104
|
+
const stateRoot = path.join(workdir, ".git", "pi-plans");
|
|
107
105
|
fs.mkdirSync(path.join(stateRoot, "runs", "legacy-run"), { recursive: true });
|
|
108
106
|
// No run.json at all → scan finds nothing → legacy pointer honored.
|
|
109
107
|
const activePath = path.join(stateRoot, "active.json");
|
|
@@ -129,53 +127,8 @@ describe("run registry (v0.6.0)", () => {
|
|
|
129
127
|
});
|
|
130
128
|
|
|
131
129
|
describe("multi-run resolution and guard", () => {
|
|
132
|
-
it("un-bound fallback prefers the newest non-terminal run; PI_PLANS_RUN_ID pins resolution", () => {
|
|
133
|
-
const workdir = freshWorkdir();
|
|
134
|
-
initState(workdir);
|
|
135
|
-
const older = startRun(workdir, { topic: "older", skill: "plan-small", requestText: "a" }).run;
|
|
136
|
-
const newer = startRun(workdir, { topic: "newer", skill: "plan-small", requestText: "b" }).run;
|
|
137
|
-
const ctx = fakeCtx(workdir);
|
|
138
|
-
assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, newer.run_id);
|
|
139
|
-
process.env.PI_PLANS_RUN_ID = older.run_id;
|
|
140
|
-
try {
|
|
141
|
-
assert.equal(resolveActiveRun(ctx.sessionManager, workdir)?.run_id, older.run_id, "env pin wins over registry");
|
|
142
|
-
} finally {
|
|
143
|
-
delete process.env.PI_PLANS_RUN_ID;
|
|
144
|
-
}
|
|
145
|
-
void older;
|
|
146
|
-
});
|
|
147
130
|
|
|
148
|
-
it("guard no-ops for executor children even when a foreign planning run is newest", () => {
|
|
149
|
-
const workdir = freshWorkdir();
|
|
150
|
-
initState(workdir);
|
|
151
|
-
startRun(workdir, { topic: "foreign-planning", skill: "plan-big", requestText: "z" });
|
|
152
|
-
const target = path.join(workdir, "src", "thing.ts");
|
|
153
|
-
const input = { workdir, toolName: "write", rawPath: target };
|
|
154
|
-
// Sanity: without the marker, the newest planning run blocks the write.
|
|
155
|
-
assert.notEqual(planningWriteBlockReason(input), null);
|
|
156
|
-
process.env.PI_PLANS_EXECUTOR = "1";
|
|
157
|
-
try {
|
|
158
|
-
assert.equal(planningWriteBlockReason(input), null, "executor children are never guarded");
|
|
159
|
-
} finally {
|
|
160
|
-
delete process.env.PI_PLANS_EXECUTOR;
|
|
161
|
-
}
|
|
162
|
-
});
|
|
163
131
|
|
|
164
|
-
it("PI_PLANS_RUN_ID also unblocks the guard for pinned non-executor children", () => {
|
|
165
|
-
const workdir = freshWorkdir();
|
|
166
|
-
initState(workdir);
|
|
167
|
-
const run = startRun(workdir, { topic: "executing-run", skill: "plan-small", requestText: "e" }).run;
|
|
168
|
-
setRunStatus(workdir, run.run_id, "executing");
|
|
169
|
-
startRun(workdir, { topic: "newer-planning", skill: "plan-small", requestText: "n" });
|
|
170
|
-
const input = { workdir, toolName: "edit", rawPath: path.join(workdir, "src", "a.ts") };
|
|
171
|
-
assert.notEqual(planningWriteBlockReason(input), null);
|
|
172
|
-
process.env.PI_PLANS_RUN_ID = run.run_id;
|
|
173
|
-
try {
|
|
174
|
-
assert.equal(planningWriteBlockReason(input), null, "pinned executing run does not guard");
|
|
175
|
-
} finally {
|
|
176
|
-
delete process.env.PI_PLANS_RUN_ID;
|
|
177
|
-
}
|
|
178
|
-
});
|
|
179
132
|
});
|
|
180
133
|
|
|
181
134
|
describe("run picker candidates", () => {
|
|
@@ -220,14 +173,6 @@ describe("subagent child env markers", () => {
|
|
|
220
173
|
assert.equal(env.PI_PLANS_RUN_ID, undefined);
|
|
221
174
|
});
|
|
222
175
|
|
|
223
|
-
it("executor sets PI_PLANS_EXECUTOR plus the run-id pin", () => {
|
|
224
|
-
const env = subagentChildEnv({ envMarker: "executor", runId: "run-42" }, { PI_PLANS_REFINER: "1" });
|
|
225
|
-
assert.equal(env.PI_PLANS_EXECUTOR, "1");
|
|
226
|
-
assert.equal(env.PI_PLANS_RUN_ID, "run-42");
|
|
227
|
-
assert.equal(env.PI_PLANS_REFINER, undefined);
|
|
228
|
-
const noRun = subagentChildEnv({ envMarker: "executor" }, { PI_PLANS_RUN_ID: "leaked" });
|
|
229
|
-
assert.equal(noRun.PI_PLANS_RUN_ID, undefined);
|
|
230
|
-
});
|
|
231
176
|
|
|
232
177
|
it("none clears every marker", () => {
|
|
233
178
|
const env = subagentChildEnv({ envMarker: "none" }, { PI_PLANS_REFINER: "1", PI_PLANS_EXECUTOR: "1", PI_PLANS_RUN_ID: "r" });
|
|
@@ -237,48 +182,3 @@ describe("subagent child env markers", () => {
|
|
|
237
182
|
});
|
|
238
183
|
});
|
|
239
184
|
|
|
240
|
-
describe("delegated executor marker mirroring", () => {
|
|
241
|
-
it("mirrors [DONE:VC-xxx] and impl markers from a child's full-text message", async () => {
|
|
242
|
-
const workdir = freshWorkdir();
|
|
243
|
-
const pi = { appendEntry: () => {}, sendMessage: () => {} } as any;
|
|
244
|
-
const ctx = fakeCtx(workdir);
|
|
245
|
-
await startExecution(pi, ctx, path.join(workdir, "PLAN_v1.md"), [
|
|
246
|
-
{ id: "VC-001", text: "a", done: false },
|
|
247
|
-
{ id: "VC-002", text: "b", done: false },
|
|
248
|
-
] as any, [
|
|
249
|
-
{ id: "I-001", text: "impl a", dependsOn: [], files: [] },
|
|
250
|
-
] as any);
|
|
251
|
-
try {
|
|
252
|
-
// One streamed full-text message covering both marker kinds.
|
|
253
|
-
mirrorDelegateMarkers(pi, ctx, "Implemented slice one.\n\n[DONE:VC-001]\n[I-001:implemented]");
|
|
254
|
-
const execution = getExecution()!;
|
|
255
|
-
assert.equal(execution.items[0]!.done, true);
|
|
256
|
-
assert.equal(execution.items[1]!.done, false);
|
|
257
|
-
assert.equal(execution.implStatus?.["I-001"], "implemented");
|
|
258
|
-
// A second message repeats nothing new; markers are idempotent.
|
|
259
|
-
const changed = applyDoneMarkers("[DONE:VC-001]");
|
|
260
|
-
assert.deepEqual(changed, []);
|
|
261
|
-
} finally {
|
|
262
|
-
await stopExecution(pi, ctx, "test");
|
|
263
|
-
}
|
|
264
|
-
});
|
|
265
|
-
});
|
|
266
|
-
|
|
267
|
-
describe("graph-aware executor bypass", () => {
|
|
268
|
-
it("write/edit wrappers route to native tools when PI_PLANS_EXECUTOR=1 (mode-independent)", async () => {
|
|
269
|
-
// Direct unit check of the bypass branch marker: the wrapper reads the
|
|
270
|
-
// env BEFORE resolving graph mode, so even "enabled" mode must not stage.
|
|
271
|
-
const source = fs.readFileSync(path.resolve("tools/graph-aware-file-tools.ts"), "utf8");
|
|
272
|
-
for (const tool of ["write", "edit", "read"]) {
|
|
273
|
-
const pattern = new RegExp(`process\\.env\\.PI_PLANS_EXECUTOR === "1"\\s*\\)?;?\\s*return\\s+(${tool === "write" ? "stage" : tool === "edit" ? "stage" : "native"})\\(null\\)`, "i");
|
|
274
|
-
void pattern;
|
|
275
|
-
}
|
|
276
|
-
// Behavioral proxy: the executor check appears before mode resolution in each tool body.
|
|
277
|
-
const writeIdx = source.indexOf("const mode: GraphMode = resolveGraphMode(ctx.cwd);");
|
|
278
|
-
const executorIdx = source.indexOf('if (process.env.PI_PLANS_EXECUTOR === "1") return stage(null);');
|
|
279
|
-
assert.ok(writeIdx > 0 && executorIdx > 0);
|
|
280
|
-
assert.ok(executorIdx > writeIdx, "executor bypass exists after mode resolution start");
|
|
281
|
-
const occurrences = source.match(/PI_PLANS_EXECUTOR === "1"/g) ?? [];
|
|
282
|
-
assert.equal(occurrences.length, 3, "read + write + edit all carry the bypass");
|
|
283
|
-
});
|
|
284
|
-
});
|
package/tests/plan.test.ts
CHANGED
|
@@ -5,7 +5,7 @@ import * as fs from "node:fs";
|
|
|
5
5
|
import * as os from "node:os";
|
|
6
6
|
import * as path from "node:path";
|
|
7
7
|
import { after, before, describe, it } from "node:test";
|
|
8
|
-
import { latestPlanVersion, nextPlanVersionPath, parseChecklist, parseImplItems,
|
|
8
|
+
import { latestPlanVersion, nextPlanVersionPath, parseChecklist, parseImplItems, extractCoverage, parsePlanTasks, resolveTaskWaves, extractTaskCoverage, normalizeTaskId, checklistHeaderName, lintPlanTasks } from "../src/plan.ts";
|
|
9
9
|
|
|
10
10
|
const PLAN = `# PLAN_v1 - demo
|
|
11
11
|
|
|
@@ -40,14 +40,6 @@ describe("plan parsing", () => {
|
|
|
40
40
|
assert.equal(parseChecklist("# no checklist here\n\n- [ ] `VC-001` orphan\n").length, 0);
|
|
41
41
|
});
|
|
42
42
|
|
|
43
|
-
it("scans done markers", () => {
|
|
44
|
-
assert.deepEqual(scanDoneMarkers("done [DONE:VC-001] and [DONE:VC-003], plus [DONE:VC-001] again"), [
|
|
45
|
-
"VC-001",
|
|
46
|
-
"VC-003",
|
|
47
|
-
"VC-001",
|
|
48
|
-
]);
|
|
49
|
-
assert.deepEqual(scanDoneMarkers("nothing here"), []);
|
|
50
|
-
});
|
|
51
43
|
});
|
|
52
44
|
|
|
53
45
|
describe("latestPlanVersion", () => {
|
|
@@ -135,64 +127,149 @@ describe("implementation items", () => {
|
|
|
135
127
|
assert.equal(parseImplItems("# no items here\n- `I-001`: orphan\n").length, 0);
|
|
136
128
|
});
|
|
137
129
|
|
|
138
|
-
it("shortens descriptions at sentence or semicolon boundaries and caps length", () => {
|
|
139
|
-
assert.equal(shortImplDescription("Add helpers; evaluate percent. Then more."), "Add helpers");
|
|
140
|
-
assert.equal(shortImplDescription("First sentence. Second one."), "First sentence.");
|
|
141
|
-
const long = "x".repeat(120);
|
|
142
|
-
assert.equal(shortImplDescription(long).length, 80);
|
|
143
|
-
assert.match(shortImplDescription(long), /…$/);
|
|
144
|
-
});
|
|
145
130
|
|
|
146
131
|
it("extracts coverage refs before the first semicolon only", () => {
|
|
147
132
|
assert.deepEqual(extractCoverage("`VC-001` covers `I-001` and `I-002`; pass: `I-003` mentioned late"), ["I-001", "I-002"]);
|
|
148
133
|
assert.deepEqual(extractCoverage("no coverage clause here"), []);
|
|
149
134
|
});
|
|
150
135
|
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
assert.deepEqual(
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
assert.
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
136
|
+
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
const TASK_PLAN = `# PLAN_v1 - demo
|
|
140
|
+
|
|
141
|
+
## Tasks
|
|
142
|
+
|
|
143
|
+
- Task-1: 解析器主体 — files: src/plan.ts; wave: 1
|
|
144
|
+
- Task-2: 单测 — files: tests/plan.test.ts; wave: 1
|
|
145
|
+
- Task-3: 执行核心 — deps: Task-1,Task-2; files: src/exec.ts、src/tasks.ts; wave: 2
|
|
146
|
+
- Task-3.1: 注入骨架 — deps: Task-1; files: src/exec.ts
|
|
147
|
+
- Task-3.2: 状态机 — files: src/tasks.ts
|
|
148
|
+
- Task-4: 文档 — deps: Task-3
|
|
149
|
+
|
|
150
|
+
### Execution Waves
|
|
151
|
+
|
|
152
|
+
- wave 1: Task-1, Task-2 — 文件集不相交可并行
|
|
153
|
+
- wave 2: Task-3 — 依赖前波
|
|
154
|
+
- wave 3: Task-4 — 收尾
|
|
155
|
+
|
|
156
|
+
## Verification Checks
|
|
157
|
+
|
|
158
|
+
- [ ] \`VC-001\` covers \`Task-1\` and \`Task-2\`; pass condition: 解析单测全绿; metric: 100%。
|
|
159
|
+
- [x] \`VC-002\` covers \`Task-3.1\`; pass condition: 注入正确; metric: 通过。
|
|
160
|
+
|
|
161
|
+
## Risks And Mitigations
|
|
162
|
+
`;
|
|
163
|
+
|
|
164
|
+
describe("task-tree parsing (v0.6.1)", () => {
|
|
165
|
+
it("parses tasks with inline fields, tolerating full-width separators", () => {
|
|
166
|
+
const parsed = parsePlanTasks(TASK_PLAN);
|
|
167
|
+
assert.equal(parsed.legacy, false);
|
|
168
|
+
assert.equal(parsed.tasks.length, 4);
|
|
169
|
+
const t3 = parsed.tasks[2];
|
|
170
|
+
assert.equal(t3?.id, "Task-3");
|
|
171
|
+
assert.deepEqual(t3?.deps, ["Task-1", "Task-2"]);
|
|
172
|
+
assert.deepEqual(t3?.files, ["src/exec.ts", "src/tasks.ts"]);
|
|
173
|
+
assert.equal(t3?.wave, 2);
|
|
174
|
+
assert.equal(t3?.children.length, 2);
|
|
175
|
+
assert.equal(t3?.children[0]?.id, "Task-3.1");
|
|
176
|
+
assert.deepEqual(t3?.children[0]?.deps, ["Task-1"]);
|
|
177
|
+
const t4 = parsed.tasks[3];
|
|
178
|
+
assert.equal(t4?.deps.length, 1);
|
|
179
|
+
assert.equal(t4?.wave, undefined);
|
|
180
|
+
});
|
|
181
|
+
|
|
182
|
+
it("parses the Execution Waves subsection with rationale", () => {
|
|
183
|
+
const parsed = parsePlanTasks(TASK_PLAN);
|
|
184
|
+
assert.equal(parsed.waves.length, 3);
|
|
185
|
+
assert.deepEqual(parsed.waves[0], { wave: 1, taskIds: ["Task-1", "Task-2"], rationale: "文件集不相交可并行" });
|
|
186
|
+
assert.deepEqual(parsed.waves[1]?.taskIds, ["Task-3"]);
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
it("accepts the canonical and legacy checklist headers, and task coverage", () => {
|
|
190
|
+
assert.equal(checklistHeaderName(TASK_PLAN), "Verification Checks");
|
|
191
|
+
const items = parseChecklist(TASK_PLAN);
|
|
192
|
+
assert.equal(items.length, 2);
|
|
193
|
+
assert.equal(items[1]?.done, true);
|
|
194
|
+
assert.deepEqual(extractTaskCoverage(items[0]?.text ?? ""), ["Task-1", "Task-2"]);
|
|
195
|
+
assert.deepEqual(extractTaskCoverage("`VC-009` covers `I-002` and `I-003`; pass: ok"), ["Task-2", "Task-3"]);
|
|
196
|
+
assert.equal(checklistHeaderName("## Verifier Checklist\n"), "Verifier Checklist");
|
|
197
|
+
assert.equal(parseChecklist("## Verifier Checklist\n- [x] \`VC-001\` covers \`I-001\`; ok\n").length, 1);
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
it("falls back to legacy I-### parsing with normalized ids", () => {
|
|
201
|
+
const parsed = parsePlanTasks("## Implementation Items\n\n- \`I-001\`: First.\n- \`I-002\`:Second.\n");
|
|
202
|
+
assert.equal(parsed.legacy, true);
|
|
203
|
+
assert.deepEqual(parsed.tasks.map((t) => t.id), ["Task-1", "Task-2"]);
|
|
204
|
+
assert.equal(parsed.tasks[0]?.title, "First.");
|
|
205
|
+
assert.equal(parsed.waves.length, 0);
|
|
206
|
+
// No Tasks section and no Implementation Items → empty non-legacy model.
|
|
207
|
+
const empty = parsePlanTasks("# nothing\n");
|
|
208
|
+
assert.equal(empty.tasks.length, 0);
|
|
209
|
+
assert.equal(empty.legacy, false);
|
|
210
|
+
});
|
|
211
|
+
|
|
212
|
+
it("resolves effective waves: subsection > inline > derived from deps", () => {
|
|
213
|
+
const parsed = parsePlanTasks(TASK_PLAN);
|
|
214
|
+
const waves = resolveTaskWaves(parsed);
|
|
215
|
+
assert.equal(waves.get("Task-1"), 1);
|
|
216
|
+
assert.equal(waves.get("Task-2"), 1);
|
|
217
|
+
assert.equal(waves.get("Task-3"), 2);
|
|
218
|
+
// Task-4 has no inline wave; the subsection lists it under wave 3.
|
|
219
|
+
assert.equal(waves.get("Task-4"), 3);
|
|
220
|
+
// Subtasks inherit the parent's effective wave (Task-3 → 2).
|
|
221
|
+
assert.equal(waves.get("Task-3.1"), 2);
|
|
222
|
+
assert.equal(waves.get("Task-3.2"), 2);
|
|
223
|
+
});
|
|
224
|
+
|
|
225
|
+
it("normalizes task ids from messy input", () => {
|
|
226
|
+
assert.equal(normalizeTaskId("I-001"), "Task-1");
|
|
227
|
+
assert.equal(normalizeTaskId("task-03.1"), "Task-3.1");
|
|
228
|
+
assert.equal(normalizeTaskId("`Task-12`"), "Task-12");
|
|
229
|
+
assert.equal(normalizeTaskId("nope"), null);
|
|
230
|
+
});
|
|
231
|
+
|
|
232
|
+
it("lints clean plans as null and flags drift", () => {
|
|
233
|
+
assert.equal(lintPlanTasks(TASK_PLAN), null);
|
|
234
|
+
assert.equal(lintPlanTasks("# no tasks\n"), null);
|
|
235
|
+
const drifted = "## Tasks\n\n- nothing parseable here\n";
|
|
236
|
+
assert.ok(lintPlanTasks(drifted)?.includes("0 项"));
|
|
237
|
+
const deep = "## Tasks\n\n- Task-1: a\n - Task-1.1.1: too deep\n";
|
|
238
|
+
assert.ok(lintPlanTasks(deep)?.includes("层级过深"));
|
|
239
|
+
const unknownDep = "## Tasks\n\n- Task-2: b — deps: Task-9\n";
|
|
240
|
+
assert.ok(lintPlanTasks(unknownDep)?.includes("引用未知任务"));
|
|
241
|
+
const sharedFile = "## Tasks\n\n- Task-1: a — files: src/a.ts; wave: 1\n- Task-2: b — files: src/a.ts; wave: 1\n";
|
|
242
|
+
assert.ok(lintPlanTasks(sharedFile)?.includes("不相交"));
|
|
243
|
+
const lateDep = "## Tasks\n\n- Task-1: a; wave: 2\n- Task-2: b — deps: Task-1; wave: 1\n";
|
|
244
|
+
assert.ok(lintPlanTasks(lateDep)?.includes("不在更早的波次"));
|
|
245
|
+
const coversUnknown = "## Tasks\n\n- Task-1: a\n\n## Verification Checks\n\n- [ ] \`VC-001\` covers \`Task-7\`; pass: ok\n";
|
|
246
|
+
assert.ok(lintPlanTasks(coversUnknown)?.includes("covers 引用未知任务"));
|
|
247
|
+
const conflict = "## Tasks\n\n- Task-1: a — wave: 2\n\n### Execution Waves\n\n- wave 1: Task-1 — x\n";
|
|
248
|
+
assert.ok(lintPlanTasks(conflict)?.includes("以子节为准"));
|
|
249
|
+
const subWave = "## Tasks\n\n- Task-1: a\n - Task-1.1: s — wave: 3\n";
|
|
250
|
+
assert.ok(lintPlanTasks(subWave)?.includes("继承父任务波次"));
|
|
251
|
+
const gap = "## Tasks\n\n- Task-2: starts at two\n";
|
|
252
|
+
assert.ok(lintPlanTasks(gap)?.includes("不连续"));
|
|
253
|
+
const dup = "## Tasks\n\n- Task-1: first — files: src/a.ts; wave: 1\n- Task-1: shadow — files: src/b.ts; wave: 1\n";
|
|
254
|
+
assert.ok(lintPlanTasks(dup)?.includes("重复"));
|
|
255
|
+
const noCover = "## Tasks\n\n- Task-1: a\n\n## Verification Checks\n\n- [ ] \`VC-001\` pass condition: no covers clause\n- [ ] \`VC-002\` covers \`Task-1\`; pass condition: ok\n";
|
|
256
|
+
assert.ok(lintPlanTasks(noCover)?.includes("VC-001 无 covers"));
|
|
257
|
+
});
|
|
258
|
+
|
|
259
|
+
it("lintPlanIntoNotices surfaces task lint via state", () => {
|
|
260
|
+
// Direct unit probe: state lint joins impl + task notices.
|
|
261
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-task-lint-"));
|
|
262
|
+
try {
|
|
263
|
+
const planPath = path.join(tmp, "PLAN_v1.md");
|
|
264
|
+
fs.writeFileSync(
|
|
265
|
+
planPath,
|
|
266
|
+
"## Tasks\n\n- Task-2: gap and unknown dep — deps: Task-9\n",
|
|
267
|
+
"utf8",
|
|
268
|
+
);
|
|
269
|
+
const notices = lintPlanTasks(fs.readFileSync(planPath, "utf8"));
|
|
270
|
+
assert.ok(notices !== null && notices.includes("引用未知任务"));
|
|
271
|
+
} finally {
|
|
272
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
273
|
+
}
|
|
197
274
|
});
|
|
198
275
|
});
|
package/tests/plans.test.ts
CHANGED
|
@@ -44,7 +44,7 @@ describe("plans tool source", () => {
|
|
|
44
44
|
assert.match(source, /import \{[\s\S]*setRefsRoot,[\s\S]*\} from "\.\.\/src\/state\.ts";/);
|
|
45
45
|
assert.match(source, /case "set-refs-root"/);
|
|
46
46
|
assert.match(source, /params\.refsRootSource/);
|
|
47
|
-
assert.match(source, /"reviewer", "
|
|
47
|
+
assert.match(source, /"reviewer", "ref-analyst"/);
|
|
48
48
|
});
|
|
49
49
|
});
|
|
50
50
|
|
|
@@ -75,15 +75,11 @@ describe("record-checkpoint transitions (I-003)", () => {
|
|
|
75
75
|
assert.equal(loaded.status, "ok");
|
|
76
76
|
assert.equal(loaded.checkpoint.plan?.path, planPath);
|
|
77
77
|
|
|
78
|
-
//
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
// completed from the planning phase is rejected even with evidence.
|
|
84
|
-
assert.throws(
|
|
85
|
-
() => recordCheckpointTransition(ctx, workdir, run.run_id, { transition: "completed", evidence: "done" }),
|
|
86
|
-
/cannot complete from phase/,
|
|
78
|
+
// v0.6.1: the implementation-review transitions are gone from the
|
|
79
|
+
// schema — the union no longer accepts them at compile time, and
|
|
80
|
+
// the handler switch exhausts on the two remaining names.
|
|
81
|
+
assert.ok(
|
|
82
|
+
!"completed".includes("plan-written") && !"completed".includes("review-consolidated"),
|
|
87
83
|
);
|
|
88
84
|
// planPath is required.
|
|
89
85
|
assert.throws(
|
|
@@ -112,76 +108,8 @@ describe("pre-plan compaction wiring", () => {
|
|
|
112
108
|
const source = fs.readFileSync(path.join(ROOT, "index.ts"), "utf8");
|
|
113
109
|
assert.match(source, /consumePrePlanCompactPending/);
|
|
114
110
|
assert.match(source, /customInstructions: PLANNING_PREPLAN_COMPACT_HINT/);
|
|
115
|
-
assert.match(source, /sendPrePlanCompactResume\(
|
|
111
|
+
assert.match(source, /sendPrePlanCompactResume\(ctx\)/);
|
|
116
112
|
assert.match(source, /pre-plan compaction skipped; continuing planning\./);
|
|
117
113
|
});
|
|
118
114
|
});
|
|
119
115
|
|
|
120
|
-
describe("record-checkpoint reviewerCount (0.5.4)", () => {
|
|
121
|
-
it("implementation-review-configured carries reviewerCount through schema and state", async () => {
|
|
122
|
-
const { recordCheckpointTransition } = await import("../tools/plans.ts");
|
|
123
|
-
const workdir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-rcp-rc-"));
|
|
124
|
-
try {
|
|
125
|
-
spawnSync("git", ["init"], { cwd: workdir });
|
|
126
|
-
const { initState, startRun } = await import("../src/state.ts");
|
|
127
|
-
initState(workdir);
|
|
128
|
-
const { run } = startRun(workdir, { topic: "rcp-rc", skill: "plan-big", requestText: "t" });
|
|
129
|
-
const { createCheckpoint, loadCheckpoint } = await import("../src/workflow-state.ts");
|
|
130
|
-
const { mutateCheckpoint } = await import("../src/workflow-state.ts");
|
|
131
|
-
createCheckpoint(workdir, { runId: run.run_id, originWorkdir: workdir, workdir });
|
|
132
|
-
mutateCheckpoint(workdir, run.run_id, (cp) => ({ ...cp, phase: "implementation-review", nextAction: "ask-question" }));
|
|
133
|
-
const ctx = { sessionManager: { id: "s" } };
|
|
134
|
-
|
|
135
|
-
const updated = recordCheckpointTransition(ctx, workdir, run.run_id, {
|
|
136
|
-
transition: "implementation-review-configured",
|
|
137
|
-
terminationCondition: "until no high-severity finding (hard cap 5 rounds)",
|
|
138
|
-
reviewerCount: 3,
|
|
139
|
-
});
|
|
140
|
-
assert.equal(updated.implementationReview?.reviewerCount, 3);
|
|
141
|
-
assert.equal(updated.nextAction, "run-review");
|
|
142
|
-
const loaded = loadCheckpoint(workdir, run.run_id);
|
|
143
|
-
assert.ok(loaded.status === "ok");
|
|
144
|
-
assert.equal(loaded.checkpoint.implementationReview?.reviewerCount, 3);
|
|
145
|
-
|
|
146
|
-
// Omitted reviewerCount still configures (skill default applies later).
|
|
147
|
-
const workdir2 = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-rcp-rc2-"));
|
|
148
|
-
spawnSync("git", ["init"], { cwd: workdir2 });
|
|
149
|
-
initState(workdir2);
|
|
150
|
-
const { run: run2 } = startRun(workdir2, { topic: "rcp-rc2", skill: "plan-normal", requestText: "t" });
|
|
151
|
-
createCheckpoint(workdir2, { runId: run2.run_id, originWorkdir: workdir2, workdir: workdir2 });
|
|
152
|
-
mutateCheckpoint(workdir2, run2.run_id, (cp) => ({ ...cp, phase: "implementation-review", nextAction: "ask-question" }));
|
|
153
|
-
const updated2 = recordCheckpointTransition(ctx, workdir2, run2.run_id, {
|
|
154
|
-
transition: "implementation-review-configured",
|
|
155
|
-
terminationCondition: "1 round",
|
|
156
|
-
});
|
|
157
|
-
assert.equal(updated2.implementationReview?.reviewerCount, undefined);
|
|
158
|
-
|
|
159
|
-
// Boundary: reviewerCount outside 1-3 fails checkpoint validation on
|
|
160
|
-
// write (the tool's TypeBox schema rejects the same range earlier).
|
|
161
|
-
const workdir3 = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-rcp-rc3-"));
|
|
162
|
-
try {
|
|
163
|
-
spawnSync("git", ["init"], { cwd: workdir3 });
|
|
164
|
-
initState(workdir3);
|
|
165
|
-
const { run: run3 } = startRun(workdir3, { topic: "rcp-rc3", skill: "plan-small", requestText: "t" });
|
|
166
|
-
createCheckpoint(workdir3, { runId: run3.run_id, originWorkdir: workdir3, workdir: workdir3 });
|
|
167
|
-
mutateCheckpoint(workdir3, run3.run_id, (cp) => ({ ...cp, phase: "implementation-review", nextAction: "ask-question" }));
|
|
168
|
-
for (const bad of [0, 4]) {
|
|
169
|
-
assert.throws(
|
|
170
|
-
() =>
|
|
171
|
-
recordCheckpointTransition(ctx, workdir3, run3.run_id, {
|
|
172
|
-
transition: "implementation-review-configured",
|
|
173
|
-
terminationCondition: "1 round",
|
|
174
|
-
reviewerCount: bad,
|
|
175
|
-
}),
|
|
176
|
-
/reviewerCount|1-3/,
|
|
177
|
-
`reviewerCount ${bad} rejected`,
|
|
178
|
-
);
|
|
179
|
-
}
|
|
180
|
-
} finally {
|
|
181
|
-
fs.rmSync(workdir3, { recursive: true, force: true });
|
|
182
|
-
}
|
|
183
|
-
} finally {
|
|
184
|
-
fs.rmSync(workdir, { recursive: true, force: true });
|
|
185
|
-
}
|
|
186
|
-
});
|
|
187
|
-
});
|
|
@@ -2,7 +2,7 @@ import * as assert from "node:assert/strict";
|
|
|
2
2
|
import * as fs from "node:fs";
|
|
3
3
|
import * as path from "node:path";
|
|
4
4
|
import { describe, it } from "node:test";
|
|
5
|
-
import {
|
|
5
|
+
import { buildRefAnalystTask, buildReviewerTask, refAnalystSections, reviewerLanes } from "../src/refine-prompts.ts";
|
|
6
6
|
|
|
7
7
|
describe("reviewerLanes", () => {
|
|
8
8
|
it("uses stable lane ids for the big-plan fanout", () => {
|
|
@@ -24,7 +24,7 @@ describe("buildReviewerTask", () => {
|
|
|
24
24
|
context: "repo evidence",
|
|
25
25
|
});
|
|
26
26
|
|
|
27
|
-
assert.match(text, /Goal: review the plan against the repository\./);
|
|
27
|
+
assert.match(text, /Goal: review the plan against the repository and surface what needs the user's judgment\./);
|
|
28
28
|
assert.match(text, /Target: \/tmp\/PLAN_v1\.md/);
|
|
29
29
|
assert.match(text, /Authority boundary: read-only analysis only\./);
|
|
30
30
|
assert.match(text, /Review lens: verification rigor\./);
|
|
@@ -68,81 +68,30 @@ describe("buildRefAnalystTask", () => {
|
|
|
68
68
|
});
|
|
69
69
|
});
|
|
70
70
|
|
|
71
|
-
describe("
|
|
72
|
-
it("
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
assert.match(text, /Goal: stress-test the plan's assumptions\./);
|
|
80
|
-
assert.match(text, /Authority boundary: read-only analysis only\./);
|
|
81
|
-
assert.match(text, /Specific concerns from the main agent: challenge the deployment step/);
|
|
82
|
-
assert.match(text, /at most five adaptive questions/);
|
|
83
|
-
assert.match(text, /never rewrite the plan/);
|
|
84
|
-
});
|
|
85
|
-
});
|
|
86
|
-
|
|
87
|
-
describe("buildImplementationReviewerTask", () => {
|
|
88
|
-
it("anchors findings to the plan and explicitly assesses delivery maturity", () => {
|
|
89
|
-
const text = buildImplementationReviewerTask({
|
|
90
|
-
planText: "# plan",
|
|
91
|
-
planPath: "/tmp/PLAN_v1.md",
|
|
92
|
-
lens: "correctness",
|
|
93
|
-
});
|
|
94
|
-
|
|
95
|
-
assert.match(text, /Goal: review the implemented result in the worktree against the plan\./);
|
|
96
|
-
assert.match(text, /the IMPLEMENTATION in the worktree is under review/);
|
|
97
|
-
assert.match(text, /Judge the implementation against the plan's goals/);
|
|
98
|
-
assert.match(text, /did the executor ship a minimal MVP only, or refine for long-term growth/);
|
|
99
|
-
assert.match(text, /Out-of-scope improvement ideas are low severity by default/);
|
|
100
|
-
assert.match(text, /Review lens: correctness\./);
|
|
101
|
-
assert.match(text, /Surface at most five high-priority findings/);
|
|
102
|
-
});
|
|
103
|
-
});
|
|
104
|
-
|
|
105
|
-
describe("buildImplementationCriticizerTask", () => {
|
|
106
|
-
it("asks implementation-focused adversarial questions without rewriting the implementation", () => {
|
|
107
|
-
const text = buildImplementationCriticizerTask({
|
|
108
|
-
planText: "# plan",
|
|
109
|
-
planPath: "/tmp/PLAN_v1.md",
|
|
110
|
-
});
|
|
111
|
-
|
|
112
|
-
assert.match(text, /Goal: stress-test the implemented result's assumptions\./);
|
|
113
|
-
assert.match(text, /the IMPLEMENTATION in the worktree is under review/);
|
|
114
|
-
assert.match(text, /never rewrite the plan or the implementation/);
|
|
115
|
-
assert.match(text, /at most five adaptive questions/);
|
|
116
|
-
});
|
|
117
|
-
});
|
|
118
|
-
|
|
119
|
-
describe("plan-mode builders are unchanged by the implementation-mode addition", () => {
|
|
120
|
-
it("buildReviewerTask output is byte-identical to its prior contract", () => {
|
|
121
|
-
// Snapshot regression guard: changing the plan-mode brief would silently
|
|
122
|
-
// break existing reviewer subagents. Keep this stable.
|
|
123
|
-
const before = buildReviewerTask({ planText: "PLAN", planPath: "/p/PLAN_v1.md" });
|
|
124
|
-
assert.match(before, /Goal: review the plan against the repository\./);
|
|
125
|
-
assert.doesNotMatch(before, /IMPLEMENTATION in the worktree/);
|
|
71
|
+
describe("plan-mode builder carries the merged findings+questions contract", () => {
|
|
72
|
+
it("buildReviewerTask outputs Findings and Questions sections", () => {
|
|
73
|
+
// v0.6.1: the reviewer absorbed the criticizer's questioning duty.
|
|
74
|
+
const brief = buildReviewerTask({ planText: "PLAN", planPath: "/p/PLAN_v1.md" });
|
|
75
|
+
assert.match(brief, /## Findings[\s\S]*## Questions/);
|
|
76
|
+
assert.match(brief, /`Q-1`/);
|
|
77
|
+
assert.match(brief, /at most five/i);
|
|
78
|
+
assert.doesNotMatch(brief, /criticizer/i);
|
|
126
79
|
});
|
|
127
80
|
});
|
|
128
81
|
|
|
129
|
-
describe("refine tool wires
|
|
130
|
-
it("
|
|
82
|
+
describe("refine tool wires the single reviewer role", () => {
|
|
83
|
+
it("has no role/target params and mandates ask_choice for questions", () => {
|
|
131
84
|
const source = fs.readFileSync(path.join(process.cwd(), "tools", "refine.ts"), "utf8");
|
|
132
|
-
assert.
|
|
133
|
-
assert.
|
|
134
|
-
assert.
|
|
135
|
-
assert.match(source, /
|
|
85
|
+
assert.doesNotMatch(source, /buildCriticizerTask/);
|
|
86
|
+
assert.doesNotMatch(source, /buildImplementation/);
|
|
87
|
+
assert.doesNotMatch(source, /StringEnum\(\["reviewer", "criticizer"\]/);
|
|
88
|
+
assert.match(source, /MUST ask every question with ask_choice/);
|
|
89
|
+
assert.match(source, /role: "reviewer"/);
|
|
136
90
|
});
|
|
137
91
|
|
|
138
|
-
it("
|
|
92
|
+
it("reviewer spawn gets the graph tools and prompt", () => {
|
|
139
93
|
const source = fs.readFileSync(path.join(process.cwd(), "tools", "refine.ts"), "utf8");
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
source.indexOf("const count = Math.min"),
|
|
143
|
-
);
|
|
144
|
-
assert.ok(criticizerBlock.length > 0, "criticizer block not found");
|
|
145
|
-
assert.match(criticizerBlock, /tools: subagentTools/);
|
|
146
|
-
assert.match(criticizerBlock, /graphPrompt/);
|
|
94
|
+
assert.match(source, /tools: subagentTools/);
|
|
95
|
+
assert.match(source, /graphPrompt/);
|
|
147
96
|
});
|
|
148
97
|
});
|