pi-plans 0.5.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (112) hide show
  1. package/CONTRIBUTING.md +126 -0
  2. package/README.md +49 -39
  3. package/agents/ref-analyst.md +7 -4
  4. package/agents/reviewer.md +12 -3
  5. package/index.ts +74 -40
  6. package/package.json +2 -1
  7. package/references/pi-planning-workflow.md +45 -58
  8. package/references/plan-artifact-template.md +71 -60
  9. package/references/state-and-config.md +63 -47
  10. package/scripts/bench/pi-adapter/pi_plans_bench.py +33 -22
  11. package/scripts/bench/pi-adapter/rpc_driver.mjs +4 -4
  12. package/scripts/run-tests.ts +12 -1
  13. package/scripts/validate.ts +22 -10
  14. package/skills/debug-and-plan/SKILL.md +4 -4
  15. package/skills/plan-big/SKILL.md +5 -5
  16. package/skills/plan-normal/SKILL.md +5 -5
  17. package/skills/plan-small/SKILL.md +5 -5
  18. package/skills/plan-with-refs/SKILL.md +8 -8
  19. package/skills/planning/SKILL.md +1 -1
  20. package/src/ask-form.ts +4 -4
  21. package/src/auditor.ts +126 -0
  22. package/src/auto-approve.ts +1 -1
  23. package/src/autocomplete.ts +19 -17
  24. package/src/code-graph/commands.ts +2 -2
  25. package/src/code-graph/community.ts +1 -1
  26. package/src/code-graph/paths.ts +1 -1
  27. package/src/code-graph/watch.ts +2 -2
  28. package/src/compaction.ts +3 -3
  29. package/src/config-command.ts +154 -76
  30. package/src/dashboard.ts +257 -0
  31. package/src/exec.ts +709 -705
  32. package/src/global-state.ts +304 -0
  33. package/src/guard.ts +16 -3
  34. package/src/messaging.ts +44 -0
  35. package/src/plan.ts +421 -112
  36. package/src/query-hook.ts +4 -4
  37. package/src/refine-prompts.ts +14 -72
  38. package/src/refine-ui-helpers.ts +24 -5
  39. package/src/refine-ui-state.ts +1 -1
  40. package/src/refine-ui.ts +1 -1
  41. package/src/resume-command.ts +40 -130
  42. package/src/resume.ts +15 -17
  43. package/src/role-panels.ts +542 -0
  44. package/src/run-context.ts +5 -4
  45. package/src/run-picker.ts +98 -0
  46. package/src/state.ts +380 -77
  47. package/src/subagent.ts +32 -1
  48. package/src/task-tool.ts +100 -0
  49. package/src/tasks.ts +189 -0
  50. package/src/thinking-levels.ts +67 -0
  51. package/src/ui-language.ts +3 -54
  52. package/src/workflow-state.ts +78 -57
  53. package/tests/analyze-refs.test.ts +35 -18
  54. package/tests/ask-choice-pros-cons.test.ts +147 -0
  55. package/tests/ask-choice-schema.test.ts +0 -12
  56. package/tests/ask-choice.test.ts +2 -49
  57. package/tests/ask-form-tool.test.ts +4 -5
  58. package/tests/ask-form.test.ts +2 -2
  59. package/tests/auditor.test.ts +111 -0
  60. package/tests/auto-approve.test.ts +7 -10
  61. package/tests/autocomplete.test.ts +8 -11
  62. package/tests/code-graph-apply-action.test.ts +2 -2
  63. package/tests/code-graph-commands.test.ts +2 -2
  64. package/tests/code-graph-index.test.ts +2 -2
  65. package/tests/code-graph-loop.e2e.test.ts +1 -1
  66. package/tests/code-graph-mutations.test.ts +1 -1
  67. package/tests/code-graph-rollback.test.ts +1 -1
  68. package/tests/code-graph-v05.test.ts +2 -2
  69. package/tests/compaction.test.ts +1 -1
  70. package/tests/config-command.test.ts +103 -100
  71. package/tests/dashboard.test.ts +268 -0
  72. package/tests/exec-lifecycle.test.ts +181 -115
  73. package/tests/exec-panel-lifecycle.test.ts +106 -251
  74. package/tests/exec.test.ts +617 -1706
  75. package/tests/execute-plan.test.ts +44 -19
  76. package/tests/extension-load.test.ts +48 -0
  77. package/tests/global-state.test.ts +371 -0
  78. package/tests/graph-aware-file-tools.test.ts +5 -5
  79. package/tests/guard.test.ts +1 -1
  80. package/tests/multi-run.test.ts +184 -0
  81. package/tests/plan.test.ts +139 -62
  82. package/tests/plans.test.ts +7 -79
  83. package/tests/refine-prompts.test.ts +20 -71
  84. package/tests/refine-resume.test.ts +27 -22
  85. package/tests/refine-ui.test.ts +6 -15
  86. package/tests/resume-lifecycle.test.ts +37 -22
  87. package/tests/resume.test.ts +43 -88
  88. package/tests/role-panels.test.ts +391 -0
  89. package/tests/run-context.test.ts +1 -1
  90. package/tests/run-ownership.test.ts +1 -1
  91. package/tests/stale-ctx.test.ts +218 -0
  92. package/tests/state.test.ts +151 -32
  93. package/tests/subagent-thinking.test.ts +65 -0
  94. package/tests/subagent-usage.test.ts +1 -1
  95. package/tests/task-tool.test.ts +61 -0
  96. package/tests/thinking-levels.test.ts +77 -0
  97. package/tests/ui-language.test.ts +2 -17
  98. package/tests/workflow-state.test.ts +17 -99
  99. package/tools/analyze-refs.ts +67 -32
  100. package/tools/ask-choice.ts +19 -49
  101. package/tools/code-graph.ts +2 -2
  102. package/tools/execute-plan.ts +63 -33
  103. package/tools/graph-aware-file-tools.ts +6 -4
  104. package/tools/plans.ts +40 -66
  105. package/tools/refine.ts +101 -164
  106. package/agents/criticizer.md +0 -18
  107. package/scripts/bench/pi-adapter/__pycache__/pi_plans_bench.cpython-312.pyc +0 -0
  108. package/src/panel.ts +0 -473
  109. package/src/termination-prompt.ts +0 -73
  110. package/tests/goal-wait.test.ts +0 -269
  111. package/tests/panel-i-zero.test.ts +0 -420
  112. package/tests/panel.test.ts +0 -355
@@ -12,13 +12,22 @@ import { initState, setRole, setLanguage, startRun, readActive } from "../src/st
12
12
  const ROOT = path.dirname(path.dirname(url.fileURLToPath(import.meta.url)));
13
13
 
14
14
  let tmpRoot: string;
15
+ let globalDir: string;
16
+ let previousGlobalDir: string | undefined;
15
17
 
16
18
  before(() => {
17
19
  tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-analyze-refs-"));
20
+ // Isolate the global reviewer config (F-002): never touch ~/.pi/pi-plans.
21
+ previousGlobalDir = process.env.PI_PLANS_GLOBAL_DIR;
22
+ globalDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-plans-global-analyze-refs-"));
23
+ process.env.PI_PLANS_GLOBAL_DIR = globalDir;
18
24
  });
19
25
 
20
26
  after(() => {
27
+ if (previousGlobalDir === undefined) delete process.env.PI_PLANS_GLOBAL_DIR;
28
+ else process.env.PI_PLANS_GLOBAL_DIR = previousGlobalDir;
21
29
  fs.rmSync(tmpRoot, { recursive: true, force: true });
30
+ fs.rmSync(globalDir, { recursive: true, force: true });
22
31
  });
23
32
 
24
33
  function mkWorkdir(name: string): string {
@@ -102,7 +111,7 @@ function subagentLines(workdir: string): Array<any> {
102
111
  .map((line) => JSON.parse(line));
103
112
  }
104
113
 
105
- describe("analyze_refs gates", () => {
114
+ describe("analyze_refs gates", () => {
106
115
  it("refuses when no pi-plans state exists", async () => {
107
116
  const workdir = mkWorkdir("gates-no-state");
108
117
  const tool = loadTool();
@@ -112,35 +121,43 @@ describe("analyze_refs gates", () => {
112
121
  );
113
122
  });
114
123
 
115
- it("refuses with role-setting guidance when reviewer mode is invalid", async () => {
116
- const workdir = mkWorkdir("gates-bad-mode");
124
+ it("refuses with embedded text guidance when the reviewer model is unconfirmed (headless)", async () => {
125
+ const workdir = mkWorkdir("gates-unconfirmed");
117
126
  initState(workdir);
118
- const configPath = path.join(workdir, ".git", "pi_plans", "config.json");
119
- const config = JSON.parse(fs.readFileSync(configPath, "utf8"));
120
- config.reviewer.mode = "bogus";
121
- fs.writeFileSync(configPath, `${JSON.stringify(config, null, "\t")}\n`, "utf8");
122
127
  const tool = loadTool();
123
128
  await assert.rejects(
124
129
  tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
125
- /reviewer role mode is missing or invalid/,
130
+ /model was never confirmed/,
126
131
  );
127
132
  });
128
133
 
129
- it("refuses current-session reviewer mode with a switch-to-delegated message", async () => {
134
+ it("ignores the reviewer mode entirely (Q-4=B): current-session proceeds once a model is confirmed", async () => {
130
135
  const workdir = mkWorkdir("gates-current-session");
131
136
  initState(workdir);
132
- setRole(workdir, { role: "reviewer", mode: "current-session" });
133
- const tool = loadTool();
134
- await assert.rejects(
135
- tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
136
- /current-session.*delegated-subagent/s,
137
+ setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "fake/model", confirmed: true });
138
+ const refDir = path.join(workdir, "refs", "solo");
139
+ fs.mkdirSync(refDir, { recursive: true });
140
+ const restore = withFakePi(
141
+ fakePiScript(
142
+ `emit({ type: "message_end", message: { role: "assistant", model: "fake/model", content: [{ type: "text", text: "OK" }] } });`,
143
+ ),
137
144
  );
145
+ const tool = loadTool();
146
+ try {
147
+ const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
148
+ assert.match(result.content[0]!.text, /mode is current-session, but analyze_refs always spawns/);
149
+ assert.match(result.content[0]!.text, /OK/);
150
+ } finally {
151
+ restore();
152
+ }
138
153
  });
139
154
 
140
- it("refuses with confirmation guidance when the reviewer model is unconfirmed", async () => {
141
- const workdir = mkWorkdir("gates-unconfirmed");
155
+ it("requires a confirmed model even in current-session mode (analysis always spawns)", async () => {
156
+ const workdir = mkWorkdir("gates-current-session-unconfirmed");
142
157
  initState(workdir);
143
- setRole(workdir, { role: "reviewer", mode: "delegated-subagent" });
158
+ // 'inherit' fully resets selector AND confirmation — the prior test's
159
+ // confirmed selector must not carry over (the global config is shared).
160
+ setRole(workdir, { role: "reviewer", mode: "current-session", modelSelector: "inherit" });
144
161
  const tool = loadTool();
145
162
  await assert.rejects(
146
163
  tool.execute("c1", { refs: [{ id: "ref-1", localPath: "." }] }, undefined, undefined, headlessCtx(workdir)),
@@ -310,7 +327,7 @@ describe("analyze_refs fanout", () => {
310
327
  const result = await tool.execute("c1", { refs: [{ id: "ref-1", localPath: refDir }] }, undefined, undefined, headlessCtx(workdir));
311
328
  assert.match(result.content[0]!.text, /### pi-plans-refs-adhoc-ref-1/);
312
329
  assert.equal(readActive(workdir), null);
313
- const ledger = path.join(workdir, ".git", "pi_plans", "runs");
330
+ const ledger = path.join(workdir, ".git", "pi-plans", "runs");
314
331
  const runs = fs.existsSync(ledger) ? fs.readdirSync(ledger) : [];
315
332
  const spawnFiles = runs.flatMap((run) =>
316
333
  fs.existsSync(path.join(ledger, run, "subagents.jsonl")) ? [fs.readFileSync(path.join(ledger, run, "subagents.jsonl"), "utf8")] : [],
@@ -0,0 +1,147 @@
1
+ /**
2
+ * Feature 5 (v0.6.0): every AI-written ask_choice option must state its
3
+ * advantage AND its drawback as '✓ <advantage> / ✗ <drawback>' in the
4
+ * configured language.
5
+ *
6
+ * The rule is SOFT GUIDANCE (deliberately — no runtime validation, no new
7
+ * schema fields), so these tests pin the *contract text* the model actually
8
+ * reads: the Option schema, the tool description, the promptGuidelines, the
9
+ * batch/single `options` arrays, the six skills, and the normative reference
10
+ * doc. They also prove the convention is actually RENDERABLE by the existing
11
+ * form renderer, so guidance can never drift away from what users see.
12
+ */
13
+
14
+ import * as assert from "node:assert/strict";
15
+ import * as fs from "node:fs";
16
+ import * as path from "node:path";
17
+ import { fileURLToPath } from "node:url";
18
+ import { describe, it } from "node:test";
19
+ import { Value } from "typebox/value";
20
+ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
21
+ import { Option, AskChoiceParams, BatchQuestionParams, registerAskChoiceTool } from "../tools/ask-choice.ts";
22
+ import { createFormState, formRender, type FormQuestion } from "../src/ask-form.ts";
23
+
24
+ const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
25
+
26
+ const PRO = "✓";
27
+ const CON = "✗";
28
+
29
+ interface ToolDef {
30
+ name: string;
31
+ description: string;
32
+ promptSnippet: string;
33
+ promptGuidelines: string[];
34
+ parameters: unknown;
35
+ execute: (id: string, params: unknown, signal: undefined, update: undefined, ctx: unknown) => Promise<unknown>;
36
+ }
37
+
38
+ function loadTool(): ToolDef {
39
+ let tool: ToolDef | undefined;
40
+ const pi = { registerTool: (definition: ToolDef) => { tool = definition; } } as unknown as ExtensionAPI;
41
+ registerAskChoiceTool(pi);
42
+ if (!tool) throw new Error("ask_choice tool not registered");
43
+ return tool;
44
+ }
45
+
46
+ /** Read a TypeBox property description out of a schema object. */
47
+ function propertyDescription(schema: unknown, property: string): string {
48
+ const properties = (schema as { properties?: Record<string, { description?: string }> }).properties;
49
+ const found = properties?.[property]?.description;
50
+ assert.equal(typeof found, "string", `expected a description on property ${property}`);
51
+ return found!;
52
+ }
53
+
54
+ const SKILL_DIRS = ["planning", "debug-and-plan", "plan-small", "plan-normal", "plan-big", "plan-with-refs"];
55
+
56
+ describe("ask_choice pros/cons contract (v0.6.0 feature 5)", () => {
57
+ it("Option.description requires both halves via the ✓ / ✗ markers", () => {
58
+ const text = propertyDescription(Option, "description");
59
+ assert.ok(text.includes(PRO), "description must show the advantage marker");
60
+ assert.ok(text.includes(CON), "description must show the drawback marker");
61
+ assert.ok(/drawback/i.test(text), "description must name the drawback half");
62
+ assert.ok(/advantage/i.test(text), "description must name the advantage half");
63
+ assert.ok(/configured language/i.test(text), "description must pin the configured language");
64
+ });
65
+
66
+ it("the registered tool description states the rule and exempts the tool-appended tails", () => {
67
+ const tool = loadTool();
68
+ assert.ok(tool.description.includes(PRO) && tool.description.includes(CON));
69
+ assert.ok(/advantage/i.test(tool.description) && /drawback/i.test(tool.description));
70
+ assert.ok(/Other/.test(tool.description) && /Auto-complete/.test(tool.description));
71
+ });
72
+
73
+ it("promptGuidelines carry the rule as an imperative, not as new schema fields", () => {
74
+ const tool = loadTool();
75
+ assert.ok(Array.isArray(tool.promptGuidelines));
76
+ const joined = tool.promptGuidelines.join("\n");
77
+ assert.ok(joined.includes(PRO) && joined.includes(CON), "a guideline must carry the markers");
78
+ assert.ok(/drawback/i.test(joined), "a guideline must name the drawback half");
79
+ // The failure mode this guards: a model inventing `pros`/`cons` keys,
80
+ // which Option rejects (additionalProperties: false).
81
+ assert.ok(
82
+ /no separate pros\/cons fields|there are no separate pros\/cons fields/i.test(joined),
83
+ "guidelines must forbid inventing pros/cons fields",
84
+ );
85
+ });
86
+
87
+ it("both the batch and single-question options arrays propagate the rule", () => {
88
+ const batch = propertyDescription(BatchQuestionParams, "options");
89
+ assert.ok(batch.includes(PRO) && batch.includes(CON));
90
+ assert.ok(/configured language/i.test(batch));
91
+
92
+ const single = propertyDescription(AskChoiceParams, "options");
93
+ assert.ok(single.includes(PRO) && single.includes(CON));
94
+ assert.ok(/configured language/i.test(single));
95
+ });
96
+
97
+ it("every skill states the per-option pros/drawbacks rule", () => {
98
+ for (const dir of SKILL_DIRS) {
99
+ const file = path.join(ROOT, "skills", dir, "SKILL.md");
100
+ const text = fs.readFileSync(file, "utf8");
101
+ assert.ok(text.includes(PRO), `${dir}: must show the advantage marker`);
102
+ assert.ok(text.includes(CON), `${dir}: must show the drawback marker`);
103
+ }
104
+ });
105
+
106
+ it("the normative workflow doc states the rule at the options clause", () => {
107
+ const text = fs.readFileSync(path.join(ROOT, "references", "pi-planning-workflow.md"), "utf8");
108
+ const optionsBullet = text.split("\n").find((line) => line.startsWith("- `options`:"));
109
+ assert.ok(optionsBullet, "the options clause must exist");
110
+ assert.ok(optionsBullet!.includes(PRO) && optionsBullet!.includes(CON));
111
+ assert.ok(/configured language/i.test(optionsBullet!));
112
+ assert.ok(/accept\/execute/.test(optionsBullet!), "the handoff must be in scope");
113
+ });
114
+
115
+ it("the convention is renderable by the existing form renderer", () => {
116
+ const question: FormQuestion = {
117
+ question: "Which storage approach?",
118
+ options: [
119
+ { label: "Reuse the existing table", description: `${PRO} no migration / ${CON} needs a backfill`, recommended: true },
120
+ { label: "Add a new table", description: `${PRO} clean isolation / ${CON} doubles write cost` },
121
+ ],
122
+ allowOther: true,
123
+ questionId: "q1",
124
+ autoComplete: true,
125
+ };
126
+ const rows = formRender(createFormState([question]), 100);
127
+ const rendered = rows.join("\n");
128
+ assert.ok(rendered.includes(PRO), "the advantage half must reach the user");
129
+ assert.ok(rendered.includes(CON), "the drawback half must reach the user");
130
+ assert.ok(rendered.includes("no migration"), "the advantage text must be shown");
131
+ assert.ok(rendered.includes("backfill"), "the drawback text must be shown");
132
+ });
133
+
134
+ it("a convention-following option still validates against the Option schema", () => {
135
+ // No new keys: the whole point of riding `description` is that the
136
+ // existing schema accepts it unchanged.
137
+ const ok = Value.Check(Option, {
138
+ label: "Reuse the existing table",
139
+ description: `${PRO} no migration / ${CON} needs a backfill`,
140
+ recommended: true,
141
+ });
142
+ assert.equal(ok, true);
143
+ // And stray pros/cons keys are still rejected loudly.
144
+ const stray = Value.Check(Option, { label: "x", pros: "a", cons: "b" });
145
+ assert.equal(stray, false);
146
+ });
147
+ });
@@ -157,18 +157,6 @@ describe("VC-2 · hardening does not wound legitimate single-question shapes", (
157
157
  assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
158
158
  });
159
159
 
160
- it("single question with trailing + autoComplete:false passes Value.Check", () => {
161
- const wellFormed = {
162
- question: "实现评审循环如何终止?",
163
- options: [
164
- { label: "goal wait:直到无未通过 VC", recommended: true },
165
- { label: "直到无高危发现(硬帽 5 轮)" },
166
- ],
167
- trailing: "auto-refine-loop",
168
- autoComplete: false,
169
- };
170
- assert.equal(Value.Check(AskChoiceParams as never, wellFormed), true);
171
- });
172
160
  });
173
161
 
174
162
  describe("VC-3 · zero-recommended batches reject loudly with zero side effects", () => {
@@ -66,41 +66,7 @@ const OPTIONS = [
66
66
  ];
67
67
 
68
68
  describe("ask_choice trailing option", () => {
69
- it("renders Auto-refine loop as the trailing option and suppresses Auto-complete", async () => {
70
- const tool = loadTool();
71
- let seenLabels: string[] = [];
72
- const ctx = makeCtx({
73
- select: async (_question, labels) => {
74
- seenLabels = labels;
75
- return labels.find((label) => label.startsWith("Auto-refine loop"));
76
- },
77
- });
78
- const result = await tool.execute("t1", {
79
- question: "Ameliorate?",
80
- options: OPTIONS,
81
- autoComplete: true, // deliberately erroneous: trailing must suppress it
82
- trailing: "auto-refine-loop",
83
- }, undefined, undefined, ctx);
84
-
85
- assert.equal(seenLabels.at(-1)?.startsWith("Auto-refine loop"), true, "Auto-refine loop must be last");
86
- assert.equal(seenLabels.some((label) => label.startsWith("Auto-complete")), false, "Auto-complete must be absent");
87
- const text = result.content[0].text as string;
88
- assert.match(text, /User selected Auto-refine loop/);
89
- assert.match(text, /until no high-severity finding \(hard cap 5 rounds\)/);
90
- assert.match(text, /goal wait: continue until no unpassed VCs remain/);
91
- assert.match(text, /refine \(role: "reviewer", target: "implementation", reviewers: <configured reviewerCount>\)/);
92
- // 0.5.4: the trailing instructions now include the reviewer-count
93
- // question with a skill-aware default (D-1/D-2) and the deferred
94
- // combined persistence (D-5).
95
- assert.match(text, /impl-review-reviewer-count/);
96
- assert.match(text, /How many concurrent reviewers should each implementation-review round use\?/);
97
- assert.match(text, /1 \(recommended\)/, "default workdir run has no active run → default 1");
98
- assert.match(text, /Do not persist yet/);
99
- assert.equal(result.details.source, "user");
100
- assert.equal(result.details.answer, "Auto-refine loop");
101
- });
102
-
103
- it("keeps Auto-complete as the trailing option without the trailing param", async () => {
69
+ it("keeps Auto-complete as the trailing option (the auto-refine trailing param was removed in v0.6.1)", async () => {
104
70
  const tool = loadTool();
105
71
  let seenLabels: string[] = [];
106
72
  const ctx = makeCtx({
@@ -117,22 +83,9 @@ describe("ask_choice trailing option", () => {
117
83
  assert.equal(seenLabels.at(-1)?.startsWith("Auto-complete"), true);
118
84
  assert.equal(seenLabels.some((label) => label.startsWith("Auto-refine loop")), false);
119
85
  });
120
-
121
- it("never auto-answers a trailing question in headless sessions", async () => {
122
- const tool = loadTool();
123
- const ctx = makeCtx({ hasUI: false, select: async () => undefined });
124
- await assert.rejects(
125
- tool.execute("t3", {
126
- question: "Ameliorate?",
127
- options: OPTIONS,
128
- autoComplete: true,
129
- trailing: "auto-refine-loop",
130
- }, undefined, undefined, ctx),
131
- /No UI available/,
132
- );
133
- });
134
86
  });
135
87
 
88
+
136
89
  describe("ask_choice panel fitting", () => {
137
90
  it("sanitizes newlines in the question and labels", () => {
138
91
  const fitted = fitAskChoicePanel("Q line1\nQ line2", [{ core: "1. a\nb", display: "1. a\nb — desc" }], 100, 30);
@@ -15,6 +15,7 @@ import { initState, recordDecision, startRun } from "../src/state.ts";
15
15
  import { createCheckpoint, loadCheckpoint } from "../src/workflow-state.ts";
16
16
  import { reconcileCheckpointWithLedger } from "../src/resume.ts";
17
17
  import { readDecisionLedger } from "../src/resume.ts";
18
+ import { setMessagingApi } from "../src/messaging.ts";
18
19
 
19
20
  type ToolDef = {
20
21
  execute: (id: string, params: any, signal: undefined, update: undefined, ctx: any) => Promise<any>;
@@ -58,7 +59,7 @@ function startActiveRun(workdir: string) {
58
59
  }
59
60
 
60
61
  function makeCtx(workdir: string, overrides: Record<string, unknown> = {}) {
61
- return {
62
+ const ctx = {
62
63
  cwd: workdir,
63
64
  hasUI: true,
64
65
  sessionManager: {},
@@ -70,6 +71,8 @@ function makeCtx(workdir: string, overrides: Record<string, unknown> = {}) {
70
71
  ...overrides,
71
72
  },
72
73
  };
74
+ setMessagingApi({ appendEntry: () => {}, sendMessage: () => {}, sendUserMessage: async () => {} });
75
+ return ctx;
73
76
  }
74
77
 
75
78
  function recordDecisionEntry(workdir: string, runId: string, entry: Record<string, unknown>): void {
@@ -139,10 +142,6 @@ describe("ask_choice batch validation (R-013b/D-025)", () => {
139
142
  () => tool.execute("t", { question: "single?", options: [{ label: "A" }], questions: [Q1, Q2] }, undefined, undefined, ctx),
140
143
  /not both/,
141
144
  );
142
- await assert.rejects(
143
- () => tool.execute("t", { questions: [Q1, Q2], trailing: "auto-refine-loop" }, undefined, undefined, ctx),
144
- /trailing/,
145
- );
146
145
  });
147
146
 
148
147
  it("rejects the canonical scope-confirm id in batches (F-003)", async () => {
@@ -37,7 +37,7 @@ describe("form state machine", () => {
37
37
  it("preselects the recommended option as cursor without marking it answered", () => {
38
38
  const state = createFormState(qs(2));
39
39
  assert.deepEqual(state.selection, [0, 0]);
40
- // goal-x semantics: a pre-positioned cursor is NOT an answer — chips
40
+ // a pre-positioned cursor is NOT an answer — chips
41
41
  // stay □ until the user presses Enter (or commits a custom answer).
42
42
  assert.equal(allAnswered(state), false);
43
43
  assert.deepEqual(state.confirmed, [false, false]);
@@ -428,7 +428,7 @@ describe("recommended marker hygiene (0.4.1)", () => {
428
428
  });
429
429
  });
430
430
 
431
- describe("themed question frame (pi-goal-x alignment, 0.4.1)", () => {
431
+ describe("themed question frame (0.4.1)", () => {
432
432
  const T: FormTheme = {
433
433
  fg: (c, t) => `⟪${c}⟩${t}⟪/⟫`,
434
434
  bg: (c, t) => `⟦${c}⟧${t}⟦/⟧`,
@@ -0,0 +1,111 @@
1
+ /** Tests for the completion auditor (v0.6.1): verdict parsing, rollback
2
+ * boundaries, skipped-pass, and no-cover exclusion. */
3
+
4
+ import * as assert from "node:assert/strict";
5
+ import { describe, it } from "node:test";
6
+ import type { CheckItem } from "../src/plan.ts";
7
+ import { buildTaskView } from "../src/tasks.ts";
8
+ import { applyAuditOutcome, buildAuditTask, parseAuditReport, presolvedCheckIds } from "../src/auditor.ts";
9
+ import { parsePlanTasks } from "../src/plan.ts";
10
+
11
+ const PLAN = `## Tasks
12
+
13
+ - Task-1: parser — files: src/a.ts; wave: 1
14
+ - Task-2: tool — files: src/b.ts; wave: 1
15
+ - Task-3: core — deps: Task-1, Task-2; files: src/c.ts; wave: 2
16
+ - Task-3.1: injection — files: src/c.ts
17
+
18
+ ## Verification Checks
19
+
20
+ - [ ] \`VC-001\` covers \`Task-1\`; pass condition: parser ok
21
+ - [ ] \`VC-002\` covers \`Task-2\` and \`Task-3\`; pass condition: tool+core ok
22
+ - [ ] \`VC-003\` covers \`Task-3.1\`; pass condition: injection ok
23
+ - [ ] \`VC-004\` covers nothing here; pass condition: orphan
24
+ `;
25
+
26
+ function checks(done: string[] = []): CheckItem[] {
27
+ const raw = PLAN.split("## Verification Checks")[1] ?? "";
28
+ const items: CheckItem[] = [];
29
+ for (const line of raw.split("\n")) {
30
+ const m = line.match(/^- \[ \] `(VC-\d+)` (.+)$/);
31
+ if (m) items.push({ id: m[1], text: m[2], done: done.includes(m[1]) });
32
+ }
33
+ return items;
34
+ }
35
+
36
+ function view(progress?: Record<string, { status: "complete" | "skipped" }>) {
37
+ return buildTaskView(parsePlanTasks(PLAN), progress);
38
+ }
39
+
40
+ describe("auditor parsing", () => {
41
+ it("parses per-check verdicts and ignores unknown ids", () => {
42
+ const report = [
43
+ "- `VC-001` — verdict: pass; evidence: src/a.ts; note: ok",
44
+ "- `VC-002` — verdict: fail; evidence: src/c.ts; note: missing",
45
+ "- `VC-999` — verdict: pass; evidence: n/a",
46
+ ].join("\n");
47
+ const { passed, failed } = parseAuditReport(report, checks());
48
+ assert.deepEqual(passed, ["VC-001"]);
49
+ assert.deepEqual(failed, ["VC-002"]);
50
+ });
51
+
52
+ it("conflicting verdicts for one check resolve to fail", () => {
53
+ const report = "- `VC-001` — verdict: pass; note: a\n- `VC-001` — verdict: fail; note: b";
54
+ const { passed, failed } = parseAuditReport(report, checks());
55
+ assert.deepEqual(passed, []);
56
+ assert.deepEqual(failed, ["VC-001"]);
57
+ });
58
+
59
+ it("builds the brief with the auditable checks only", () => {
60
+ const task = buildAuditTask("/tmp/PLAN_v1.md", checks(), view(), 1);
61
+ assert.match(task, /read-only/i);
62
+ assert.match(task, /`VC-001`/);
63
+ assert.doesNotMatch(task, /`VC-999`/);
64
+ });
65
+ });
66
+
67
+ describe("audit rollback boundaries", () => {
68
+ it("failed multi-cover check rolls back every covered task", () => {
69
+ const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
70
+ const checklist = checks();
71
+ const outcome = applyAuditOutcome(checklist, tasks, 1, ["VC-001"], ["VC-002"], "report");
72
+ assert.deepEqual(outcome.rolledBack.sort(), ["Task-2", "Task-3", "Task-3.1"]);
73
+ // Passed check stays done; its task stays closed.
74
+ assert.equal(checks().length, 4);
75
+ assert.equal(tasks.find((t) => t.id === "Task-1")?.status, "complete");
76
+ assert.equal(tasks.find((t) => t.id === "Task-2")?.status, "pending");
77
+ });
78
+
79
+ it("covering the parent cascades to subtasks even when the parent alone is listed", () => {
80
+ const tasks = view({ "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
81
+ const outcome = applyAuditOutcome(checks(), tasks, 1, [], ["VC-002"], "report");
82
+ assert.ok(outcome.rolledBack.includes("Task-3.1"), "subtask reopens with the parent");
83
+ });
84
+
85
+ it("skipped tasks reopen too (skip state cleared)", () => {
86
+ const tasks = view({ "Task-2": { status: "skipped" } });
87
+ const outcome = applyAuditOutcome(checks(), tasks, 1, [], ["VC-002"], "report");
88
+ assert.ok(outcome.rolledBack.includes("Task-2"));
89
+ assert.equal(tasks.find((t) => t.id === "Task-2")?.skipReason, undefined);
90
+ });
91
+
92
+ it("checks with no task coverage are excluded from presolved audit", () => {
93
+ const tasks = view({ "Task-1": { status: "complete" }, "Task-2": { status: "complete" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
94
+ const presolved = presolvedCheckIds(checks(), tasks);
95
+ assert.ok(!presolved.includes("VC-004"), "no-cover check is never audited");
96
+ });
97
+
98
+ it("mixed coverage (skipped + complete siblings) resolves as skipped-pass (I-007)", () => {
99
+ const tasks = view({ "Task-2": { status: "skipped" }, "Task-3": { status: "complete" }, "Task-3.1": { status: "complete" } });
100
+ const presolved = presolvedCheckIds(checks(), tasks);
101
+ assert.ok(presolved.includes("VC-002"), "skipped + complete siblings presolve");
102
+ assert.ok(!presolved.includes("VC-003"), "pure complete coverage goes to the auditor");
103
+ });
104
+
105
+ it("all-skipped coverage resolves as skipped-pass", () => {
106
+ const tasks = view({ "Task-3": { status: "skipped" }, "Task-3.1": { status: "skipped" } });
107
+ const presolved = presolvedCheckIds(checks(), tasks);
108
+ assert.ok(presolved.includes("VC-003"));
109
+ assert.ok(!presolved.includes("VC-002"), "mixed coverage needs the auditor");
110
+ });
111
+ });
@@ -10,7 +10,7 @@ import * as path from "node:path";
10
10
  import { after, before, describe, it } from "node:test";
11
11
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
12
12
  import { registerAskChoiceTool } from "../tools/ask-choice.ts";
13
- import { executeHandoff, setCurrentApi } from "../tools/execute-plan.ts";
13
+ import { executeHandoff } from "../tools/execute-plan.ts";
14
14
  import {
15
15
  AUTO_APPROVE_ENV,
16
16
  assertAutoApprovable,
@@ -18,6 +18,7 @@ import {
18
18
  isExternalStateQuestion,
19
19
  } from "../src/auto-approve.ts";
20
20
  import { getExecution, stopExecution } from "../src/exec.ts";
21
+ import { setMessagingApi } from "../src/messaging.ts";
21
22
 
22
23
  type ToolDef = {
23
24
  execute: (id: string, params: any, signal: undefined, update: undefined, ctx: any) => Promise<any>;
@@ -55,7 +56,7 @@ function loadTool(): ToolDef {
55
56
  function makeCtx(opts: { hasUI?: boolean } = {}): any {
56
57
  selectCalls = 0;
57
58
  confirmCalls = 0;
58
- return {
59
+ const ctx = {
59
60
  cwd: root,
60
61
  hasUI: opts.hasUI ?? false,
61
62
  sessionManager: {},
@@ -74,6 +75,8 @@ function makeCtx(opts: { hasUI?: boolean } = {}): any {
74
75
  theme: { fg: (_kind: string, text: string) => text, bold: (text: string) => text },
75
76
  },
76
77
  };
78
+ setMessagingApi({ appendEntry: () => {}, sendMessage: () => {}, sendUserMessage: async () => {} });
79
+ return ctx;
77
80
  }
78
81
 
79
82
  const LIFECYCLE_OPTIONS = [
@@ -241,15 +244,9 @@ describe("execute_handoff under auto-approve", () => {
241
244
  });
242
245
 
243
246
  function makeExecMocks() {
244
- const pi = {
245
- appendEntry: () => {},
246
- sendMessage: () => {},
247
- setModel: async () => true,
248
- } as unknown as ExtensionAPI;
249
- setCurrentApi(pi);
250
247
  const ctx = makeCtx({ hasUI: false });
251
248
  ctx.cwd = root;
252
- return { pi, ctx };
249
+ return { ctx };
253
250
  }
254
251
 
255
252
  it("approves headlessly with an [auto-approve] annotated result and no confirm call", async () => {
@@ -261,7 +258,7 @@ describe("execute_handoff under auto-approve", () => {
261
258
  assert.match(outcome.message, /^\[auto-approve\] Execution approved/);
262
259
  assert.equal(confirmCalls, 0, "no confirm prompt may open");
263
260
  } finally {
264
- if (getExecution()) await stopExecution({ appendEntry: () => {}, sendMessage: () => {} } as any, ctx, "cleanup");
261
+ if (getExecution()) await stopExecution(ctx, "cleanup");
265
262
  setEnv(undefined);
266
263
  }
267
264
  });
@@ -14,9 +14,9 @@ import {
14
14
  recordAskChoice,
15
15
  registerAutoCompleteTurnHandlers,
16
16
  restoreAutoCompleteFromSession,
17
- setAutoCompleteApi,
18
17
  } from "../src/autocomplete.ts";
19
18
  import { initState, setRunStatus, startRun } from "../src/state.ts";
19
+ import { setMessagingApi } from "../src/messaging.ts";
20
20
 
21
21
  let root: string;
22
22
 
@@ -46,7 +46,6 @@ before(() => {
46
46
  });
47
47
 
48
48
  after(() => {
49
- setAutoCompleteApi(null);
50
49
  fs.rmSync(root, { recursive: true, force: true });
51
50
  });
52
51
 
@@ -54,8 +53,8 @@ describe("Auto-complete state", () => {
54
53
  it("enables only for the active planning run and restores only while planning", () => {
55
54
  const { workdir, runId } = makeRun("restore");
56
55
  const entries: any[] = [];
57
- setAutoCompleteApi({ appendEntry: (customType: string, data: unknown) => entries.push({ type: "custom", customType, data }) } as any);
58
56
  const ctx = makeContext(workdir);
57
+ setMessagingApi({ appendEntry: (customType: string, data: unknown) => entries.push({ type: "custom", customType, data }), sendMessage: () => {}, sendUserMessage: async () => {} });
59
58
 
60
59
  assert.equal(enableAutoComplete(ctx), true);
61
60
  assert.equal(autoCompleteStatus(ctx), "enabled");
@@ -79,8 +78,8 @@ describe("Auto-complete state", () => {
79
78
  it("clears explicitly and marks plan writes as a continuation boundary", () => {
80
79
  const { workdir } = makeRun("clear");
81
80
  const entries: any[] = [];
82
- setAutoCompleteApi({ appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }) } as any);
83
81
  const ctx = makeContext(workdir);
82
+ setMessagingApi({ appendEntry: (customType: string, data: unknown) => entries.push({ customType, data }), sendMessage: () => {}, sendUserMessage: async () => {} });
84
83
  enableAutoComplete(ctx);
85
84
  markPlanWritten(ctx);
86
85
  recordAskChoice(ctx, true);
@@ -98,12 +97,10 @@ describe("Auto-complete continuation", () => {
98
97
  const handlers = new Map<string, Function[]>();
99
98
  const pi: any = {
100
99
  on: (name: string, handler: Function) => handlers.set(name, [...(handlers.get(name) ?? []), handler]),
101
- appendEntry: () => {},
102
- sendUserMessage: async (content: string, options: unknown) => sent.push({ content, options }),
103
100
  };
104
- setAutoCompleteApi(pi);
105
101
  registerAutoCompleteTurnHandlers(pi);
106
102
  const ctx = makeContext(workdir, session);
103
+ setMessagingApi({ appendEntry: () => {}, sendMessage: () => {}, sendUserMessage: async (content: string, options: unknown) => sent.push({ content, options }) });
107
104
  enableAutoComplete(ctx);
108
105
  await handlers.get("turn_start")?.[0]?.({}, ctx);
109
106
  recordAskChoice(ctx, true);
@@ -132,10 +129,10 @@ describe("ask_choice Auto-complete wiring", () => {
132
129
  assert.match(source, /isAutoCompleteEnabled/);
133
130
  assert.match(source, /recordAskChoice\(ctx, true\)/);
134
131
  assert.match(source, /autoComplete && selected\.startsWith\("Auto-complete"\)/);
135
- // Trailing option: Auto-refine loop replaces Auto-complete and is suppressed in headless sessions.
136
- assert.match(source, /trailing === undefined/);
137
- assert.match(source, /AUTO_REFINE_LOOP_LABEL/);
138
- assert.match(source, /trailing && selected\.startsWith\("Auto-refine loop"\)/);
132
+ // v0.6.1: the auto-refine trailing param was removed with the
133
+ // implementation-review loop.
134
+ assert.doesNotMatch(source, /auto-refine-loop/);
135
+ assert.doesNotMatch(source, /AUTO_REFINE_LOOP_LABEL/);
139
136
  const indexSource = fs.readFileSync(path.join(process.cwd(), "index.ts"), "utf8");
140
137
  assert.match(indexSource, /plans-autocomplete-stop/);
141
138
  assert.match(indexSource, /autoCompleteStatus\(ctx\)/);
@@ -51,9 +51,9 @@ async function setupIndexedRepo(): Promise<{ root: string; store: Store; cleanup
51
51
  tsx: makeBackend("tsx", ParserCtor, runtime.runtime.parser.tsx),
52
52
  python: new PythonBackend(ParserCtor, runtime.runtime.parser.python),
53
53
  };
54
- fs.mkdirSync(path.join(root, ".git", "pi_plans"), { recursive: true });
54
+ fs.mkdirSync(path.join(root, ".git", "pi-plans"), { recursive: true });
55
55
  const store = new Store(
56
- { dbPath: path.join(root, ".git", "pi_plans", "code_graph.db"), worktreeRoot: root, gitCommonDir: path.join(root, ".git") },
56
+ { dbPath: path.join(root, ".git", "pi-plans", "code_graph.db"), worktreeRoot: root, gitCommonDir: path.join(root, ".git") },
57
57
  runtime.runtime.sqlite,
58
58
  );
59
59
  runIndex({ store, worktreeRoot: root, parsers, reindex: false });
@@ -31,7 +31,7 @@ function initRepo(): { root: string; cleanup: () => void } {
31
31
  fs.writeFileSync(path.join(root, "helper.ts"), "export function twice(n: number): number { return n * 2; }\n");
32
32
  git(root, ["add", "-A"]);
33
33
  git(root, ["commit", "-m", "init"]);
34
- fs.mkdirSync(path.join(root, ".git", "pi_plans"), { recursive: true });
34
+ fs.mkdirSync(path.join(root, ".git", "pi-plans"), { recursive: true });
35
35
  const canonicalRoot = fs.realpathSync(root);
36
36
  return {
37
37
  root: canonicalRoot,
@@ -63,7 +63,7 @@ async function loadParsers() {
63
63
 
64
64
  function openStore(root: string, sqlite: typeof import("node:sqlite")): Store {
65
65
  const canonicalRoot = fs.realpathSync(root);
66
- const dbPath = path.join(canonicalRoot, ".git", "pi_plans", "code_graph.db");
66
+ const dbPath = path.join(canonicalRoot, ".git", "pi-plans", "code_graph.db");
67
67
  fs.mkdirSync(path.dirname(dbPath), { recursive: true });
68
68
  return new Store({ dbPath, worktreeRoot: canonicalRoot, gitCommonDir: path.join(canonicalRoot, ".git") }, sqlite);
69
69
  }