@fyeeme/pi-goal 1.0.2 → 1.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/index.ts +82 -8
- package/package.json +1 -1
- package/src/evaluator.ts +22 -4
- package/src/runtime.ts +21 -0
- package/src/tool.ts +7 -0
- package/test/evaluator.test.ts +32 -1
- package/test/index.test.ts +90 -2
- package/test/runtime.test.ts +18 -0
- package/test/tool.test.ts +26 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
### Added
|
|
6
|
+
|
|
7
|
+
- Objective length cap (`MAX_OBJECTIVE_CHARS`, 4,000 code points) enforced by the runtime on `create` and `replace` — the objective is re-injected into context on every continuation and passed to the evaluator subprocess, so an oversized one silently burns budget every turn; the error guides long instructions into a referenced file. Counting is code-point-based, not UTF-16 units. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- Print/json/rpc modes could never create a goal: `session_start` removed the `goal` tool from the active set when no goal existed (omp sdk.ts parity), but non-interactive modes have no `/goal` or `/guided-goal` command to re-arm it — leaving the model unable to start a goal at all (found via live goal-mode testing). The tool is now only removed in TUI mode.
|
|
12
|
+
- The run that creates/resumes a goal via the tool never saw the goal context prompt: `before_agent_start` injection only fires on the next agent run, and the command-path steer was not wired to the tool path (found via live goal-mode testing). The tool now fires an `onActivated` hook so the host injects the goal context into the current run as a steer.
|
|
13
|
+
- The evaluator subprocess ran with full extension discovery: any other installed goal extension (with an active-goal system prompt) would pollute the judge that is supposed to be independent, and the startup overhead contributed to live 300s timeouts. It now runs lean (`--no-extensions --no-skills`), the default timeout is raised to 600s, and `GOAL_EVALUATOR_TIMEOUT_MS` overrides it.
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- Context hygiene: hidden goal messages no longer accumulate in the LLM's view. A `context` handler keeps only the newest `goal-mode-context` and `goal-budget-limit` messages plus the newest `goal-continuation` stamped for the currently active goal id (`details.goalId`); stale ones — including all continuations once no goal is active — are dropped from the model's view, not from the transcript. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
18
|
+
- Run-error handling: when a run ends with an assistant `stopReason: "error"`, the active goal now pauses (persisted) instead of letting the continuation loop fire into a likely retry loop, with a classified notice — provider usage/rate/quota/limit errors read differently from generic faults. Abort behavior is unchanged. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
19
|
+
|
|
5
20
|
## [1.0.2] - 2026-09-16
|
|
6
21
|
|
|
7
22
|
### Changed
|
package/index.ts
CHANGED
|
@@ -97,6 +97,7 @@ const goalModeContextPrompt = readFileSync(
|
|
|
97
97
|
interface EntryMessageLike {
|
|
98
98
|
role?: string;
|
|
99
99
|
stopReason?: string;
|
|
100
|
+
errorMessage?: string;
|
|
100
101
|
usage?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined;
|
|
101
102
|
}
|
|
102
103
|
|
|
@@ -340,6 +341,23 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
340
341
|
await exitGoalMode({ reason: "paused" });
|
|
341
342
|
}
|
|
342
343
|
|
|
344
|
+
/** A provider/agent error ended the run (borrowed from mitsuhiko/agent-stuff
|
|
345
|
+
* goal.ts): pause the active goal instead of letting scheduleContinuation
|
|
346
|
+
* fire into a likely retry loop, and classify the failure for the user.
|
|
347
|
+
* Usage/rate/quota errors read differently from generic faults. */
|
|
348
|
+
async function pauseOnRunError(messages: unknown[]): Promise<void> {
|
|
349
|
+
if (!goalState?.enabled || goalState.goal.status !== "active") return;
|
|
350
|
+
const errorMessage = lastAssistantErrorMessage(messages) ?? "";
|
|
351
|
+
const usageLimited = /\b(usage|rate|quota|limit)\b/i.test(errorMessage);
|
|
352
|
+
await runtime.pauseGoal();
|
|
353
|
+
notify(
|
|
354
|
+
usageLimited
|
|
355
|
+
? "Goal paused: the last turn hit provider usage/rate limits. /goal resume when ready."
|
|
356
|
+
: "Goal paused: the last turn ended with an error. /goal resume to continue.",
|
|
357
|
+
"error",
|
|
358
|
+
);
|
|
359
|
+
}
|
|
360
|
+
|
|
343
361
|
async function dropGoal(): Promise<void> {
|
|
344
362
|
if (!goalState) {
|
|
345
363
|
notify("No goal to drop.", "warning");
|
|
@@ -408,7 +426,9 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
408
426
|
if (!prompt) return;
|
|
409
427
|
continuationInFlight = true;
|
|
410
428
|
pi.sendMessage(
|
|
411
|
-
|
|
429
|
+
// details.goalId keys context pruning (see the "context" handler below):
|
|
430
|
+
// only the newest continuation of the CURRENTLY active goal survives.
|
|
431
|
+
{ customType: "goal-continuation", content: prompt, display: false, details: { goalId: goalState.goal.id } },
|
|
412
432
|
{ triggerTurn: true, deliverAs: "followUp" },
|
|
413
433
|
);
|
|
414
434
|
}
|
|
@@ -425,6 +445,11 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
425
445
|
// fall back to the process cwd.
|
|
426
446
|
runEvaluator: (request, opts) =>
|
|
427
447
|
runGoalEvaluator(request, { cwd: opts.cwd ?? currentCtx?.cwd ?? process.cwd(), signal: opts.signal }),
|
|
448
|
+
// Mid-run activation (tool create/resume): inject the goal context into
|
|
449
|
+
// the CURRENT run as a steer — before_agent_start only covers the next run.
|
|
450
|
+
onActivated: async () => {
|
|
451
|
+
if (currentCtx && !currentCtx.isIdle()) await sendGoalModeContext("steer");
|
|
452
|
+
},
|
|
428
453
|
};
|
|
429
454
|
pi.registerTool(createGoalTool(goalToolDeps));
|
|
430
455
|
|
|
@@ -501,11 +526,16 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
501
526
|
const restored = restoreGoalFromEntries(branch);
|
|
502
527
|
if (!restored) {
|
|
503
528
|
goalState = undefined;
|
|
504
|
-
// omp sdk.ts excludes the goal tool from the initial set; mirror that
|
|
505
|
-
//
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
529
|
+
// omp sdk.ts excludes the goal tool from the initial set; mirror that in
|
|
530
|
+
// TUI where /goal and /guided-goal re-arm it. In non-interactive modes
|
|
531
|
+
// (print/json/rpc) no slash command exists, so removing the tool would
|
|
532
|
+
// leave the model unable to create a goal at all (found via live
|
|
533
|
+
// goal-mode testing): keep it active there.
|
|
534
|
+
if (ctx.mode === "tui") {
|
|
535
|
+
const active = pi.getActiveTools();
|
|
536
|
+
if (active.includes("goal")) {
|
|
537
|
+
pi.setActiveTools(active.filter((name) => name !== "goal"));
|
|
538
|
+
}
|
|
509
539
|
}
|
|
510
540
|
updateStatus();
|
|
511
541
|
return;
|
|
@@ -563,15 +593,51 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
563
593
|
};
|
|
564
594
|
});
|
|
565
595
|
|
|
596
|
+
// Context hygiene (borrowed from mitsuhiko/agent-stuff goal.ts): hidden goal
|
|
597
|
+
// messages otherwise accumulate one per run/continuation and burn context
|
|
598
|
+
// every LLM call. Keep only the newest goal-mode-context and
|
|
599
|
+
// goal-budget-limit, plus the newest continuation stamped for the currently
|
|
600
|
+
// active goal id; stale ones (including all continuations once no goal is
|
|
601
|
+
// active) are dropped from the model's view, not from the transcript.
|
|
602
|
+
pi.on("context", (event) => {
|
|
603
|
+
const activeGoalId = goalState?.enabled && goalState.goal.status === "active" ? goalState.goal.id : undefined;
|
|
604
|
+
let lastContext = -1;
|
|
605
|
+
let lastBudget = -1;
|
|
606
|
+
let lastContinuation = -1;
|
|
607
|
+
for (let i = 0; i < event.messages.length; i++) {
|
|
608
|
+
const msg = event.messages[i] as
|
|
609
|
+
| { role?: string; customType?: string; details?: { goalId?: string } }
|
|
610
|
+
| undefined;
|
|
611
|
+
if (msg?.role !== "custom") continue;
|
|
612
|
+
if (msg.customType === "goal-mode-context") lastContext = i;
|
|
613
|
+
else if (msg.customType === "goal-budget-limit") lastBudget = i;
|
|
614
|
+
else if (msg.customType === "goal-continuation" && activeGoalId !== undefined && msg.details?.goalId === activeGoalId) {
|
|
615
|
+
lastContinuation = i;
|
|
616
|
+
}
|
|
617
|
+
}
|
|
618
|
+
if (lastContext === -1 && lastBudget === -1 && lastContinuation === -1) return undefined;
|
|
619
|
+
return {
|
|
620
|
+
messages: event.messages.filter((message, index) => {
|
|
621
|
+
const msg = message as { role?: string; customType?: string } | undefined;
|
|
622
|
+
if (msg?.role !== "custom") return true;
|
|
623
|
+
if (msg.customType === "goal-mode-context") return index === lastContext;
|
|
624
|
+
if (msg.customType === "goal-budget-limit") return index === lastBudget;
|
|
625
|
+
if (msg.customType === "goal-continuation") return index === lastContinuation;
|
|
626
|
+
return true;
|
|
627
|
+
}),
|
|
628
|
+
};
|
|
629
|
+
});
|
|
630
|
+
|
|
566
631
|
pi.on("agent_end", async (event, ctx) => {
|
|
567
632
|
currentCtx = ctx;
|
|
568
633
|
// omp separates onAgentEnd (session) from continuation scheduling
|
|
569
634
|
// (interactive-mode); both subscribe to the same end-of-run moment.
|
|
570
|
-
const
|
|
571
|
-
if (aborted) {
|
|
635
|
+
const stopReason = lastAssistantStopReason(event.messages);
|
|
636
|
+
if (stopReason === "aborted") {
|
|
572
637
|
await runtime.onTaskAborted({ reason: "interrupted" });
|
|
573
638
|
} else {
|
|
574
639
|
await runtime.onAgentEnd({ currentUsage: currentUsage(ctx) });
|
|
640
|
+
if (stopReason === "error") await pauseOnRunError(event.messages);
|
|
575
641
|
}
|
|
576
642
|
|
|
577
643
|
if (continuationInFlight) {
|
|
@@ -665,6 +731,14 @@ function lastAssistantStopReason(messages: unknown[]): string | undefined {
|
|
|
665
731
|
return undefined;
|
|
666
732
|
}
|
|
667
733
|
|
|
734
|
+
function lastAssistantErrorMessage(messages: unknown[]): string | undefined {
|
|
735
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
736
|
+
const message = messages[i] as EntryMessageLike | undefined;
|
|
737
|
+
if (message?.role === "assistant") return message.errorMessage;
|
|
738
|
+
}
|
|
739
|
+
return undefined;
|
|
740
|
+
}
|
|
741
|
+
|
|
668
742
|
export type { GoalRuntimeHost } from "./src/runtime.ts";
|
|
669
743
|
// Re-exported for consumers that compose the pieces directly (tests, tools).
|
|
670
744
|
export { GoalRuntime } from "./src/runtime.ts";
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@fyeeme/pi-goal",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.3",
|
|
4
4
|
"description": "Goal mode for pi \u2014 one persistent autonomous objective looped until verified success: goal tool (create/get/complete/resume/drop), token/time budget accounting with budget-limit steering, automatic continuation turns, /goal and /guided-goal commands, and a goal_updated event bus for other extensions.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
package/src/evaluator.ts
CHANGED
|
@@ -34,8 +34,16 @@ const evaluatorCompletePrompt = readFileSync(path.join(promptsDir, "evaluator-co
|
|
|
34
34
|
const evaluatorImpossiblePrompt = readFileSync(path.join(promptsDir, "evaluator-impossible.md"), "utf8");
|
|
35
35
|
|
|
36
36
|
/** Hard cap on one evaluator run — the grounded check may run test suites,
|
|
37
|
-
* so this is deliberately generous; anything longer has failed.
|
|
38
|
-
|
|
37
|
+
* so this is deliberately generous; anything longer has failed. Each LLM
|
|
38
|
+
* call on slow providers can take 20s+, and the evaluator is a multi-step
|
|
39
|
+
* agent (inspect repo, run checks, emit JSON), so the default needs real
|
|
40
|
+
* headroom. Override with GOAL_EVALUATOR_TIMEOUT_MS. */
|
|
41
|
+
export const EVALUATOR_TIMEOUT_MS = 600_000;
|
|
42
|
+
|
|
43
|
+
function evaluatorTimeout(): number {
|
|
44
|
+
const raw = Number.parseInt(process.env.GOAL_EVALUATOR_TIMEOUT_MS ?? "", 10);
|
|
45
|
+
return Number.isInteger(raw) && raw > 0 ? raw : EVALUATOR_TIMEOUT_MS;
|
|
46
|
+
}
|
|
39
47
|
|
|
40
48
|
/** Argv-safety caps (kernel MAX_ARG_STRLEN ≈ 128KB; these keep the prompt
|
|
41
49
|
* argument far under it even for verbose objectives/audits). */
|
|
@@ -118,6 +126,16 @@ export function buildEvaluatorPrompt(request: GoalEvaluatorRequest): string {
|
|
|
118
126
|
});
|
|
119
127
|
}
|
|
120
128
|
|
|
129
|
+
/** Build the evaluator argv: a one-shot headless run. Lean flags matter here:
|
|
130
|
+
* without them the nested pi loads the user's full extension set — including
|
|
131
|
+
* any OTHER goal extension, whose active-goal system prompt would pollute the
|
|
132
|
+
* very evaluator that is supposed to judge the goal independently — and its
|
|
133
|
+
* startup cost pushes long verification runs over the timeout (observed
|
|
134
|
+
* live: 300s timeout hit while the evaluator was still working). */
|
|
135
|
+
function evaluatorInvocation(prompt: string): { command: string; args: string[] } {
|
|
136
|
+
return getPiInvocation(["-p", "--no-session", "--no-extensions", "--no-skills", prompt]);
|
|
137
|
+
}
|
|
138
|
+
|
|
121
139
|
/** Extract the first balanced JSON object from evaluator output. Handles the
|
|
122
140
|
* clean case, code-fenced output, and prose-wrapped JSON; anything else is
|
|
123
141
|
* unparseable. Never throws. */
|
|
@@ -181,10 +199,10 @@ export async function runGoalEvaluator(
|
|
|
181
199
|
): Promise<GoalEvaluatorOutcome> {
|
|
182
200
|
const prompt = buildEvaluatorPrompt(request);
|
|
183
201
|
const spawn = opts.spawn ?? defaultSpawn;
|
|
184
|
-
const timeoutMs = opts.timeoutMs ??
|
|
202
|
+
const timeoutMs = opts.timeoutMs ?? evaluatorTimeout();
|
|
185
203
|
let stdout: string;
|
|
186
204
|
try {
|
|
187
|
-
stdout = await spawn(
|
|
205
|
+
stdout = await spawn(evaluatorInvocation(prompt), {
|
|
188
206
|
cwd: opts.cwd,
|
|
189
207
|
timeoutMs,
|
|
190
208
|
signal: opts.signal,
|
package/src/runtime.ts
CHANGED
|
@@ -128,6 +128,25 @@ export function completionBudgetReport(goal: Goal): string | null {
|
|
|
128
128
|
return `Goal achieved. Report final budget usage to the user: ${parts.join("; ")}.`;
|
|
129
129
|
}
|
|
130
130
|
|
|
131
|
+
/** Objective length cap. The objective is re-injected into context on every
|
|
132
|
+
* continuation and passed to the evaluator subprocess, so an oversized one
|
|
133
|
+
* silently burns budget every turn. Borrowed from mitsuhiko/agent-stuff
|
|
134
|
+
* goal.ts (MAX_OBJECTIVE_CHARS). */
|
|
135
|
+
export const MAX_OBJECTIVE_CHARS = 4_000;
|
|
136
|
+
|
|
137
|
+
function charCount(value: string): number {
|
|
138
|
+
return [...value].length;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function validateObjective(objective: string): void {
|
|
142
|
+
const count = charCount(objective);
|
|
143
|
+
if (count > MAX_OBJECTIVE_CHARS) {
|
|
144
|
+
throw new Error(
|
|
145
|
+
`Goal objective is too long: ${count.toLocaleString()} characters. Limit: ${MAX_OBJECTIVE_CHARS.toLocaleString()} characters. Put longer instructions in a file and reference that file in the objective.`,
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
131
150
|
function validateTokenBudget(tokenBudget: number | undefined): void {
|
|
132
151
|
if (tokenBudget !== undefined && (!Number.isInteger(tokenBudget) || tokenBudget <= 0)) {
|
|
133
152
|
throw new Error("goal token_budget must be a positive integer when provided");
|
|
@@ -408,6 +427,7 @@ export class GoalRuntime {
|
|
|
408
427
|
async createGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
|
|
409
428
|
const objective = input.objective.trim();
|
|
410
429
|
if (!objective) throw new Error("objective is required when op=create");
|
|
430
|
+
validateObjective(objective);
|
|
411
431
|
validateTokenBudget(input.tokenBudget);
|
|
412
432
|
return await this.#withAccounting(async () => {
|
|
413
433
|
const existing = this.#host.getState();
|
|
@@ -425,6 +445,7 @@ export class GoalRuntime {
|
|
|
425
445
|
async replaceGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
|
|
426
446
|
const objective = input.objective.trim();
|
|
427
447
|
if (!objective) throw new Error("objective is required when op=replace");
|
|
448
|
+
validateObjective(objective);
|
|
428
449
|
validateTokenBudget(input.tokenBudget);
|
|
429
450
|
return await this.#withAccounting(async () => {
|
|
430
451
|
const existing = this.#host.getState();
|
package/src/tool.ts
CHANGED
|
@@ -127,6 +127,11 @@ export interface GoalToolDeps {
|
|
|
127
127
|
/** Independent completion/impossibility evaluator (src/evaluator.ts).
|
|
128
128
|
* Gates `complete` and adjudicates `impossible`. */
|
|
129
129
|
runEvaluator: GoalEvaluatorFn;
|
|
130
|
+
/** Called after the tool ACTIVATES a goal mid-run (create/resume). The
|
|
131
|
+
* before_agent_start injection only fires on the next agent run, so
|
|
132
|
+
* without this hook the run that created the goal never sees the goal
|
|
133
|
+
* context prompt at all (observed live in print mode). */
|
|
134
|
+
onActivated?: () => Promise<void>;
|
|
130
135
|
}
|
|
131
136
|
|
|
132
137
|
function describeEvaluatorVerdict(verdict: string): string {
|
|
@@ -172,12 +177,14 @@ export function createGoalTool(deps: GoalToolDeps): ToolDefinition<typeof GoalPa
|
|
|
172
177
|
if (params.op === "create") {
|
|
173
178
|
const created = await runtime.createGoal(validateCreateParams(params));
|
|
174
179
|
response = buildGoalToolResponse(created.goal);
|
|
180
|
+
await deps.onActivated?.();
|
|
175
181
|
} else if (params.op === "get") {
|
|
176
182
|
const state = deps.getState();
|
|
177
183
|
response = buildGoalToolResponse(state?.goal ?? null);
|
|
178
184
|
} else if (params.op === "resume") {
|
|
179
185
|
const resumed = await runtime.resumeGoal();
|
|
180
186
|
response = buildGoalToolResponse(resumed.goal);
|
|
187
|
+
await deps.onActivated?.();
|
|
181
188
|
} else if (params.op === "drop") {
|
|
182
189
|
const dropped = await runtime.dropGoal();
|
|
183
190
|
response = buildGoalToolResponse(dropped ?? null);
|
package/test/evaluator.test.ts
CHANGED
|
@@ -15,7 +15,6 @@ import {
|
|
|
15
15
|
runGoalEvaluator,
|
|
16
16
|
type EvaluatorSpawn,
|
|
17
17
|
} from "../src/evaluator.ts";
|
|
18
|
-
|
|
19
18
|
function spawnReturning(stdout: string): { spawn: EvaluatorSpawn; calls: { args: string[]; cwd: string }[] } {
|
|
20
19
|
const calls: { args: string[]; cwd: string }[] = [];
|
|
21
20
|
return {
|
|
@@ -31,6 +30,38 @@ const COMPLETE_REQUEST = { mode: "complete" as const, objective: "Ship it", clai
|
|
|
31
30
|
const IMPOSSIBLE_REQUEST = { mode: "impossible" as const, objective: "Ship it", claim: "no network access" };
|
|
32
31
|
const RUN_OPTS = { cwd: "/tmp/repo" };
|
|
33
32
|
|
|
33
|
+
describe("evaluator subprocess invocation", () => {
|
|
34
|
+
it("runs lean: no extensions, no skills (a nested goal extension would pollute the judge)", async () => {
|
|
35
|
+
const run = spawnReturning('{"ok": true}');
|
|
36
|
+
await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: run.spawn });
|
|
37
|
+
const { args } = run.calls[0]!;
|
|
38
|
+
expect(args).toContain("--no-extensions");
|
|
39
|
+
expect(args).toContain("--no-skills");
|
|
40
|
+
// Prompt stays last (argv-safety caps rely on it).
|
|
41
|
+
expect(args.indexOf("--no-skills")).toBeLessThan(args.length - 1);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
it("honors GOAL_EVALUATOR_TIMEOUT_MS over the default", async () => {
|
|
45
|
+
const run = spawnReturning('{"ok": true}');
|
|
46
|
+
const timeouts: number[] = [];
|
|
47
|
+
const previous = process.env.GOAL_EVALUATOR_TIMEOUT_MS;
|
|
48
|
+
process.env.GOAL_EVALUATOR_TIMEOUT_MS = "12345";
|
|
49
|
+
try {
|
|
50
|
+
await runGoalEvaluator(COMPLETE_REQUEST, {
|
|
51
|
+
...RUN_OPTS,
|
|
52
|
+
spawn: async (invocation, opts) => {
|
|
53
|
+
timeouts.push(opts.timeoutMs);
|
|
54
|
+
return run.spawn(invocation, opts);
|
|
55
|
+
},
|
|
56
|
+
});
|
|
57
|
+
} finally {
|
|
58
|
+
if (previous === undefined) delete process.env.GOAL_EVALUATOR_TIMEOUT_MS;
|
|
59
|
+
else process.env.GOAL_EVALUATOR_TIMEOUT_MS = previous;
|
|
60
|
+
}
|
|
61
|
+
expect(timeouts).toEqual([12_345]);
|
|
62
|
+
});
|
|
63
|
+
});
|
|
64
|
+
|
|
34
65
|
describe("extractJsonObject", () => {
|
|
35
66
|
it("parses a clean JSON object", () => {
|
|
36
67
|
expect(extractJsonObject('{"ok": true, "reason": "tests pass"}')).toEqual({
|
package/test/index.test.ts
CHANGED
|
@@ -36,6 +36,7 @@ interface FakeHost {
|
|
|
36
36
|
customType: string;
|
|
37
37
|
content: string;
|
|
38
38
|
display: boolean;
|
|
39
|
+
details?: unknown;
|
|
39
40
|
options?: { deliverAs?: string; triggerTurn?: boolean };
|
|
40
41
|
}>;
|
|
41
42
|
sentUserMessages: Array<{ content: string; options?: { deliverAs?: string } }>;
|
|
@@ -114,7 +115,10 @@ function fakeHost(): FakeHost {
|
|
|
114
115
|
return host;
|
|
115
116
|
}
|
|
116
117
|
|
|
117
|
-
function createContext(
|
|
118
|
+
function createContext(
|
|
119
|
+
host: FakeHost,
|
|
120
|
+
overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean; mode?: string } = {},
|
|
121
|
+
) {
|
|
118
122
|
return {
|
|
119
123
|
ui: {
|
|
120
124
|
notify: (text: string) => {
|
|
@@ -130,7 +134,7 @@ function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending
|
|
|
130
134
|
editor: async () => undefined,
|
|
131
135
|
},
|
|
132
136
|
hasUI: true,
|
|
133
|
-
mode: "tui",
|
|
137
|
+
mode: overrides.mode ?? "tui",
|
|
134
138
|
cwd: "/tmp",
|
|
135
139
|
isIdle: () => overrides.idle ?? true,
|
|
136
140
|
hasPendingMessages: () => overrides.pending ?? false,
|
|
@@ -222,6 +226,14 @@ describe("pi-goal extension wiring", () => {
|
|
|
222
226
|
expect(host.activeTools).toContain("read");
|
|
223
227
|
});
|
|
224
228
|
|
|
229
|
+
it("session_start keeps the goal tool in non-interactive modes (no slash commands there)", async () => {
|
|
230
|
+
const host = fakeHost();
|
|
231
|
+
defaultExport(host.pi);
|
|
232
|
+
const ctx = createContext(host, { mode: "print" });
|
|
233
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" }, ctx);
|
|
234
|
+
expect(host.activeTools).toContain("goal");
|
|
235
|
+
});
|
|
236
|
+
|
|
225
237
|
it("session_start restores a persisted active goal, re-adds the tool, then pauses it (omp onThreadResumed)", async () => {
|
|
226
238
|
const host = fakeHost();
|
|
227
239
|
defaultExport(host.pi);
|
|
@@ -295,6 +307,9 @@ describe("pi-goal extension wiring", () => {
|
|
|
295
307
|
expect(continuation).toBeDefined();
|
|
296
308
|
expect(continuation?.display).toBe(false);
|
|
297
309
|
expect(continuation?.options).toMatchObject({ triggerTurn: true, deliverAs: "followUp" });
|
|
310
|
+
// details.goalId keys the context-pruning handler.
|
|
311
|
+
const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
|
|
312
|
+
expect(continuation?.details).toMatchObject({ goalId });
|
|
298
313
|
});
|
|
299
314
|
|
|
300
315
|
it("suppresses the next continuation when a continuation turn produced no tool calls", async () => {
|
|
@@ -632,4 +647,77 @@ describe("pi-goal extension wiring", () => {
|
|
|
632
647
|
expect(component.text).toContain("6K (no budget) tokens");
|
|
633
648
|
expect(component.text).toContain("20s");
|
|
634
649
|
});
|
|
650
|
+
|
|
651
|
+
it("pauses the goal instead of continuing when the run ends with a provider error", async () => {
|
|
652
|
+
const host = fakeHost();
|
|
653
|
+
defaultExport(host.pi);
|
|
654
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
655
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
656
|
+
host.sentMessages.length = 0;
|
|
657
|
+
|
|
658
|
+
await fire(host, "agent_start", { type: "agent_start" });
|
|
659
|
+
await fire(host, "agent_end", {
|
|
660
|
+
type: "agent_end",
|
|
661
|
+
messages: [
|
|
662
|
+
{ role: "assistant", stopReason: "error", errorMessage: "429 rate limit exceeded", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
663
|
+
],
|
|
664
|
+
});
|
|
665
|
+
|
|
666
|
+
// Paused and persisted, with a classified notice; no continuation queued.
|
|
667
|
+
const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
|
|
668
|
+
expect(pauseEntry?.data).toMatchObject({ enabled: false, goal: { status: "paused" } });
|
|
669
|
+
expect(host.notifications.some((n) => n.includes("rate limits"))).toBe(true);
|
|
670
|
+
expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
|
|
671
|
+
});
|
|
672
|
+
|
|
673
|
+
it("pauses with a generic notice on non-usage errors", async () => {
|
|
674
|
+
const host = fakeHost();
|
|
675
|
+
defaultExport(host.pi);
|
|
676
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
677
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
678
|
+
|
|
679
|
+
await fire(host, "agent_start", { type: "agent_start" });
|
|
680
|
+
await fire(host, "agent_end", {
|
|
681
|
+
type: "agent_end",
|
|
682
|
+
messages: [
|
|
683
|
+
{ role: "assistant", stopReason: "error", errorMessage: "socket hang up", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
684
|
+
],
|
|
685
|
+
});
|
|
686
|
+
|
|
687
|
+
expect(host.notifications.some((n) => n.includes("ended with an error"))).toBe(true);
|
|
688
|
+
});
|
|
689
|
+
|
|
690
|
+
it("context pruning keeps only the newest goal messages and drops stale continuations", async () => {
|
|
691
|
+
const host = fakeHost();
|
|
692
|
+
defaultExport(host.pi);
|
|
693
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
694
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
695
|
+
const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
|
|
696
|
+
|
|
697
|
+
const custom = (customType: string, details?: unknown) => ({ role: "custom", customType, display: false, details });
|
|
698
|
+
const messages = [
|
|
699
|
+
{ role: "user", content: "hi" },
|
|
700
|
+
custom("goal-mode-context"), // stale context
|
|
701
|
+
custom("goal-continuation", { goalId: "old-goal" }), // stale goal id
|
|
702
|
+
custom("goal-budget-limit"),
|
|
703
|
+
custom("goal-mode-context"), // newest context: kept
|
|
704
|
+
custom("goal-continuation", { goalId }), // newest for the active goal: kept
|
|
705
|
+
{ role: "assistant", content: "working" },
|
|
706
|
+
custom("goal-continuation", { goalId: "old-goal" }), // another stale: dropped
|
|
707
|
+
];
|
|
708
|
+
|
|
709
|
+
const handler = host.handlers.get("context")?.[0]!;
|
|
710
|
+
const result = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
|
|
711
|
+
const kept = result.messages.filter((m) => (m as { role?: string }).role === "custom");
|
|
712
|
+
expect(kept.map((m) => (m as { customType: string }).customType)).toEqual([
|
|
713
|
+
"goal-budget-limit",
|
|
714
|
+
"goal-mode-context",
|
|
715
|
+
"goal-continuation",
|
|
716
|
+
]);
|
|
717
|
+
|
|
718
|
+
// Once no goal is active, every continuation is dropped.
|
|
719
|
+
await host.commands.goal!.handler("drop", createContext(host));
|
|
720
|
+
const result2 = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
|
|
721
|
+
expect(result2.messages.filter((m) => (m as { customType?: string }).customType === "goal-continuation")).toHaveLength(0);
|
|
722
|
+
});
|
|
635
723
|
});
|
package/test/runtime.test.ts
CHANGED
|
@@ -468,4 +468,22 @@ describe("goal runtime", () => {
|
|
|
468
468
|
const second = await harness.runtime.createGoal({ objective: "Two" });
|
|
469
469
|
expect(first.goal.id).not.toBe(second.goal.id);
|
|
470
470
|
});
|
|
471
|
+
|
|
472
|
+
it("rejects objectives beyond the character cap on create and replace", async () => {
|
|
473
|
+
const harness = createHarness();
|
|
474
|
+
const oversized = "x".repeat(4_001);
|
|
475
|
+
await expect(harness.runtime.createGoal({ objective: oversized })).rejects.toThrow(/too long.*4,000/s);
|
|
476
|
+
// Boundary: exactly at the cap is fine.
|
|
477
|
+
await harness.runtime.createGoal({ objective: "x".repeat(4_000) });
|
|
478
|
+
// Replace path enforces the same cap.
|
|
479
|
+
await expect(harness.runtime.replaceGoal({ objective: oversized })).rejects.toThrow(/too long/);
|
|
480
|
+
});
|
|
481
|
+
|
|
482
|
+
it("counts characters as code points, not UTF-16 units", async () => {
|
|
483
|
+
const harness = createHarness();
|
|
484
|
+
// Astral emoji: 1 code point, 2 UTF-16 units. 2_001 emoji is 4_002 UTF-16
|
|
485
|
+
// units (a naive .length cap would reject it) but 2_001 code points: kept.
|
|
486
|
+
await expect(harness.runtime.createGoal({ objective: "😀".repeat(2_001) })).resolves.toBeDefined();
|
|
487
|
+
await expect(harness.runtime.createGoal({ objective: "😀".repeat(4_001) })).rejects.toThrow(/too long/);
|
|
488
|
+
});
|
|
471
489
|
});
|
package/test/tool.test.ts
CHANGED
|
@@ -90,8 +90,14 @@ function stubEvaluator(outcome: GoalEvaluatorOutcome) {
|
|
|
90
90
|
function createTestTool(
|
|
91
91
|
harness: ReturnType<typeof createRuntimeHarness>,
|
|
92
92
|
evaluator: (request: GoalEvaluatorRequest, opts: { cwd?: string }) => Promise<GoalEvaluatorOutcome>,
|
|
93
|
+
onActivated?: () => Promise<void>,
|
|
93
94
|
) {
|
|
94
|
-
return createGoalTool({
|
|
95
|
+
return createGoalTool({
|
|
96
|
+
getRuntime: () => harness.runtime,
|
|
97
|
+
getState: harness.getState,
|
|
98
|
+
runEvaluator: evaluator,
|
|
99
|
+
onActivated,
|
|
100
|
+
});
|
|
95
101
|
}
|
|
96
102
|
|
|
97
103
|
// ---------------------------------------------------------------------------
|
|
@@ -99,6 +105,25 @@ function createTestTool(
|
|
|
99
105
|
// ---------------------------------------------------------------------------
|
|
100
106
|
|
|
101
107
|
describe("goal tool", () => {
|
|
108
|
+
it("onActivated hook fires on create and resume (mid-run context injection)", async () => {
|
|
109
|
+
const harness = createRuntimeHarness();
|
|
110
|
+
const activations: string[] = [];
|
|
111
|
+
const tool = createTestTool(
|
|
112
|
+
harness,
|
|
113
|
+
async () => ({ status: "confirmed", reason: "ok" }),
|
|
114
|
+
async () => {
|
|
115
|
+
activations.push("activated");
|
|
116
|
+
},
|
|
117
|
+
);
|
|
118
|
+
await executeTool(tool, { op: "create", objective: "Ship it" });
|
|
119
|
+
expect(activations).toHaveLength(1);
|
|
120
|
+
await executeTool(tool, { op: "resume" });
|
|
121
|
+
expect(activations).toHaveLength(2);
|
|
122
|
+
// get does not re-activate.
|
|
123
|
+
await executeTool(tool, { op: "get" });
|
|
124
|
+
expect(activations).toHaveLength(2);
|
|
125
|
+
});
|
|
126
|
+
|
|
102
127
|
it("create starts a goal and returns the objective/status text", async () => {
|
|
103
128
|
const harness = createRuntimeHarness();
|
|
104
129
|
const tool = createTestTool(harness, async () => ({ status: "confirmed", reason: "verified" }));
|