@fyeeme/pi-goal 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +43 -0
- package/LICENSE +21 -0
- package/README.md +59 -0
- package/index.ts +651 -0
- package/package.json +65 -0
- package/src/commands.ts +325 -0
- package/src/evaluator.ts +218 -0
- package/src/format.ts +35 -0
- package/src/prompts/evaluator-complete.md +26 -0
- package/src/prompts/evaluator-impossible.md +20 -0
- package/src/prompts/goal-budget-limit.md +15 -0
- package/src/prompts/goal-continuation.md +30 -0
- package/src/prompts/goal-mode-active.md +23 -0
- package/src/prompts/goal-mode-context.md +4 -0
- package/src/prompts/goal-todo-context.md +12 -0
- package/src/prompts/goal.md +11 -0
- package/src/prompts/guided-goal-interview.md +37 -0
- package/src/restore.ts +92 -0
- package/src/runtime.ts +565 -0
- package/src/state.ts +124 -0
- package/src/template.ts +154 -0
- package/src/todo-bridge.ts +136 -0
- package/src/tool.ts +557 -0
- package/test/evaluator.test.ts +186 -0
- package/test/index.test.ts +557 -0
- package/test/omp-alignment.test.ts +975 -0
- package/test/runtime.test.ts +471 -0
- package/test/template.test.ts +89 -0
- package/test/tool.test.ts +528 -0
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* pi-goal — independent evaluator tests.
|
|
3
|
+
*
|
|
4
|
+
* Pins the CC 2.1.261-absorbed contract: strict JSON verdict parsing with the
|
|
5
|
+
* "insufficient evidence = not met" default direction, grounded prompt
|
|
6
|
+
* assembly (objective + claim embedded as escaped data), and the
|
|
7
|
+
* unavailable-fallback behavior for spawn/timeout/parse failures. The spawn
|
|
8
|
+
* is injected — no subprocess runs in tests.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { describe, expect, it } from "vitest";
|
|
12
|
+
import {
|
|
13
|
+
buildEvaluatorPrompt,
|
|
14
|
+
extractJsonObject,
|
|
15
|
+
runGoalEvaluator,
|
|
16
|
+
type EvaluatorSpawn,
|
|
17
|
+
} from "../src/evaluator.ts";
|
|
18
|
+
|
|
19
|
+
function spawnReturning(stdout: string): { spawn: EvaluatorSpawn; calls: { args: string[]; cwd: string }[] } {
|
|
20
|
+
const calls: { args: string[]; cwd: string }[] = [];
|
|
21
|
+
return {
|
|
22
|
+
calls,
|
|
23
|
+
spawn: async (invocation, opts) => {
|
|
24
|
+
calls.push({ args: invocation.args, cwd: opts.cwd });
|
|
25
|
+
return stdout;
|
|
26
|
+
},
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
const COMPLETE_REQUEST = { mode: "complete" as const, objective: "Ship it", claim: "ran the tests; all green" };
|
|
31
|
+
const IMPOSSIBLE_REQUEST = { mode: "impossible" as const, objective: "Ship it", claim: "no network access" };
|
|
32
|
+
const RUN_OPTS = { cwd: "/tmp/repo" };
|
|
33
|
+
|
|
34
|
+
describe("extractJsonObject", () => {
|
|
35
|
+
it("parses a clean JSON object", () => {
|
|
36
|
+
expect(extractJsonObject('{"ok": true, "reason": "tests pass"}')).toEqual({
|
|
37
|
+
ok: true,
|
|
38
|
+
reason: "tests pass",
|
|
39
|
+
});
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
it("parses fenced and prose-wrapped JSON", () => {
|
|
43
|
+
const fenced = '```json\n{"ok": false, "reason": "missing"}\n```';
|
|
44
|
+
expect(extractJsonObject(fenced)).toEqual({ ok: false, reason: "missing" });
|
|
45
|
+
const wrapped = 'The verdict is: {"ok": true, "reason": "evidence"} — done.';
|
|
46
|
+
expect(extractJsonObject(wrapped)).toEqual({ ok: true, reason: "evidence" });
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
it("returns undefined for garbage or non-object JSON", () => {
|
|
50
|
+
expect(extractJsonObject("no json here")).toBeUndefined();
|
|
51
|
+
expect(extractJsonObject('[1, 2, 3]')).toBeUndefined();
|
|
52
|
+
expect(extractJsonObject("")).toBeUndefined();
|
|
53
|
+
});
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
describe("runGoalEvaluator — complete mode", () => {
|
|
57
|
+
it("maps ok:true to confirmed and ok:false to refuted", async () => {
|
|
58
|
+
const confirmed = spawnReturning('{"ok": true, "reason": "npm test exits 0"}');
|
|
59
|
+
const confirmedOutcome = await runGoalEvaluator(COMPLETE_REQUEST, {
|
|
60
|
+
...RUN_OPTS,
|
|
61
|
+
spawn: confirmed.spawn,
|
|
62
|
+
});
|
|
63
|
+
expect(confirmedOutcome).toEqual({ status: "confirmed", reason: "npm test exits 0" });
|
|
64
|
+
|
|
65
|
+
const refuted = spawnReturning('{"ok": false, "reason": "build fails: TS2345"}');
|
|
66
|
+
const refutedOutcome = await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: refuted.spawn });
|
|
67
|
+
expect(refutedOutcome).toEqual({ status: "refuted", reason: "build fails: TS2345" });
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it("embeds the escaped objective and claim in the prompt and runs pi non-interactively", async () => {
|
|
71
|
+
const run = spawnReturning('{"ok": true}');
|
|
72
|
+
await runGoalEvaluator(
|
|
73
|
+
{ mode: "complete", objective: "Fix <the> bug & ship", claim: "evidence with <tags>" },
|
|
74
|
+
{ ...RUN_OPTS, spawn: run.spawn },
|
|
75
|
+
);
|
|
76
|
+
expect(run.calls).toHaveLength(1);
|
|
77
|
+
const { args, cwd } = run.calls[0]!;
|
|
78
|
+
expect(cwd).toBe("/tmp/repo");
|
|
79
|
+
const prompt = args.at(-1)!;
|
|
80
|
+
// getPiInvocation may prefix the current script (node <script> ...);
|
|
81
|
+
// the flags must sit immediately before the prompt either way.
|
|
82
|
+
const flagIndex = args.indexOf("-p");
|
|
83
|
+
expect(flagIndex).toBeGreaterThan(-1);
|
|
84
|
+
expect(args.slice(flagIndex, flagIndex + 2)).toEqual(["-p", "--no-session"]);
|
|
85
|
+
expect(args.at(-1)).toBe(prompt);
|
|
86
|
+
expect(prompt).toContain("<the> bug & ship");
|
|
87
|
+
expect(prompt).toContain("evidence with <tags>");
|
|
88
|
+
expect(prompt).toContain("independent completion evaluator");
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
it("treats out-of-contract verdicts as unavailable", async () => {
|
|
92
|
+
const noOk = spawnReturning('{"verdict": "maybe"}');
|
|
93
|
+
expect(await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: noOk.spawn })).toMatchObject({
|
|
94
|
+
status: "unavailable",
|
|
95
|
+
});
|
|
96
|
+
const notObject = spawnReturning('"ok"');
|
|
97
|
+
expect(await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: notObject.spawn })).toMatchObject({
|
|
98
|
+
status: "unavailable",
|
|
99
|
+
});
|
|
100
|
+
});
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
describe("runGoalEvaluator — impossible mode", () => {
|
|
104
|
+
it("maps impossible:true/false onto confirmed/refuted", async () => {
|
|
105
|
+
const confirmed = spawnReturning('{"impossible": true, "reason": "self-contradictory condition"}');
|
|
106
|
+
expect(await runGoalEvaluator(IMPOSSIBLE_REQUEST, { ...RUN_OPTS, spawn: confirmed.spawn })).toEqual({
|
|
107
|
+
status: "confirmed",
|
|
108
|
+
reason: "self-contradictory condition",
|
|
109
|
+
});
|
|
110
|
+
const refuted = spawnReturning('{"impossible": false, "reason": "found a workable path"}');
|
|
111
|
+
expect(await runGoalEvaluator(IMPOSSIBLE_REQUEST, { ...RUN_OPTS, spawn: refuted.spawn })).toEqual({
|
|
112
|
+
status: "refuted",
|
|
113
|
+
reason: "found a workable path",
|
|
114
|
+
});
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
it("uses the impossibility prompt template", async () => {
|
|
118
|
+
const run = spawnReturning('{"impossible": false}');
|
|
119
|
+
await runGoalEvaluator(IMPOSSIBLE_REQUEST, { ...RUN_OPTS, spawn: run.spawn });
|
|
120
|
+
expect(run.calls[0]!.args.at(-1)).toContain("IMPOSSIBLE");
|
|
121
|
+
});
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
describe("runGoalEvaluator — failure paths", () => {
|
|
125
|
+
it("spawn errors are unavailable; caller aborts rethrow", async () => {
|
|
126
|
+
const failing: EvaluatorSpawn = async () => {
|
|
127
|
+
throw new Error("spawn ENOENT");
|
|
128
|
+
};
|
|
129
|
+
expect(await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: failing })).toEqual({
|
|
130
|
+
status: "unavailable",
|
|
131
|
+
detail: "spawn ENOENT",
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
const abortController = new AbortController();
|
|
135
|
+
const aborting: EvaluatorSpawn = async () => {
|
|
136
|
+
throw new Error("aborted");
|
|
137
|
+
};
|
|
138
|
+
abortController.abort();
|
|
139
|
+
await expect(
|
|
140
|
+
runGoalEvaluator(COMPLETE_REQUEST, {
|
|
141
|
+
...RUN_OPTS,
|
|
142
|
+
spawn: aborting,
|
|
143
|
+
signal: abortController.signal,
|
|
144
|
+
}),
|
|
145
|
+
).rejects.toThrow("goal evaluation aborted");
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
it("timeouts surface as unavailable with the timeout detail", async () => {
|
|
149
|
+
const timingOut: EvaluatorSpawn = async () => {
|
|
150
|
+
const err = new Error("Command timed out") as Error & { killed: boolean };
|
|
151
|
+
err.killed = true;
|
|
152
|
+
throw err;
|
|
153
|
+
};
|
|
154
|
+
const outcome = await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: timingOut });
|
|
155
|
+
expect(outcome).toMatchObject({ status: "unavailable" });
|
|
156
|
+
if (outcome.status === "unavailable") {
|
|
157
|
+
expect(outcome.detail).toContain("timed out");
|
|
158
|
+
}
|
|
159
|
+
});
|
|
160
|
+
|
|
161
|
+
it("empty and unparseable output are unavailable", async () => {
|
|
162
|
+
const empty = spawnReturning(" \n");
|
|
163
|
+
expect(await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: empty.spawn })).toMatchObject({
|
|
164
|
+
status: "unavailable",
|
|
165
|
+
detail: "evaluator produced no output",
|
|
166
|
+
});
|
|
167
|
+
const garbage = spawnReturning("I could not decide.");
|
|
168
|
+
const outcome = await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: garbage.spawn });
|
|
169
|
+
expect(outcome).toMatchObject({ status: "unavailable" });
|
|
170
|
+
if (outcome.status === "unavailable") {
|
|
171
|
+
expect(outcome.detail).toContain("unparseable evaluator output");
|
|
172
|
+
}
|
|
173
|
+
});
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
describe("buildEvaluatorPrompt", () => {
|
|
177
|
+
it("truncates oversized objectives and claims with a marker", () => {
|
|
178
|
+
const prompt = buildEvaluatorPrompt({
|
|
179
|
+
mode: "complete",
|
|
180
|
+
objective: "x".repeat(5000),
|
|
181
|
+
claim: "y".repeat(9000),
|
|
182
|
+
});
|
|
183
|
+
expect(prompt).toContain("…[truncated]");
|
|
184
|
+
expect(prompt.length).toBeLessThan(20_000);
|
|
185
|
+
});
|
|
186
|
+
});
|