@fyeeme/pi-goal 1.0.2 → 1.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/index.ts +201 -18
- package/package.json +10 -10
- package/src/evaluator.ts +22 -4
- package/src/runtime.ts +21 -0
- package/src/tool.ts +7 -0
- package/test/evaluator.test.ts +32 -1
- package/test/index.test.ts +243 -28
- package/test/runtime.test.ts +18 -0
- package/test/tool.test.ts +26 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,24 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [1.0.4] - 2026-10-01
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Objective length cap (`MAX_OBJECTIVE_CHARS`, 4,000 code points) enforced by the runtime on `create` and `replace` — the objective is re-injected into context on every continuation and passed to the evaluator subprocess, so an oversized one silently burns budget every turn; the error guides long instructions into a referenced file. Counting is code-point-based, not UTF-16 units. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- Print/json/rpc modes could never create a goal: `session_start` removed the `goal` tool from the active set when no goal existed (omp sdk.ts parity), but non-interactive modes have no `/goal` or `/guided-goal` command to re-arm it — leaving the model unable to start a goal at all (found via live goal-mode testing). The tool is now only removed in TUI mode.
|
|
14
|
+
- The run that creates/resumes a goal via the tool never saw the goal context prompt: `before_agent_start` injection only fires on the next agent run, and the command-path steer was not wired to the tool path (found via live goal-mode testing). The tool now fires an `onActivated` hook so the host injects the goal context into the current run as a steer.
|
|
15
|
+
- The evaluator subprocess ran with full extension discovery: any other installed goal extension (with an active-goal system prompt) would pollute the judge that is supposed to be independent, and the startup overhead contributed to live 300s timeouts. It now runs lean (`--no-extensions --no-skills`), the default timeout is raised to 600s, and `GOAL_EVALUATOR_TIMEOUT_MS` overrides it.
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
|
|
19
|
+
- Context hygiene: hidden goal messages no longer accumulate in the LLM's view. A `context` handler keeps only the newest `goal-mode-context` and `goal-budget-limit` messages plus the newest `goal-continuation` stamped for the currently active goal id (`details.goalId`); stale ones — including all continuations once no goal is active — are dropped from the model's view, not from the transcript. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
20
|
+
- Run-error handling: when a run ends with an assistant `stopReason: "error"`, the active goal now pauses (persisted) instead of letting the continuation loop fire into a likely retry loop, with a classified notice — provider usage/rate/quota/limit errors read differently from generic faults. Abort behavior is unchanged. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
|
|
21
|
+
- Peer dependency floor raised to `@earendil-works/pi-coding-agent >= 0.99.0`; dev toolchain pinned to 0.99.2 (typecheck and tests pass against 0.99.2 unchanged).
|
|
22
|
+
|
|
5
23
|
## [1.0.2] - 2026-09-16
|
|
6
24
|
|
|
7
25
|
### Changed
|
package/index.ts
CHANGED
|
@@ -27,12 +27,18 @@
|
|
|
27
27
|
* goal_updated session event → pi.events.emit("goal_updated")
|
|
28
28
|
* sendHiddenMessage (budget steer) → pi.sendMessage display:false
|
|
29
29
|
* prompt-time prependMessages goal context
|
|
30
|
-
* →
|
|
31
|
-
* injection (
|
|
30
|
+
* → context_with_system per-request
|
|
31
|
+
* injection (0.87.0): survives
|
|
32
|
+
* compaction, supersedes transcript
|
|
33
|
+
* copies instead of accumulating
|
|
32
34
|
* #scheduleGoalContinuation 800ms TUI timer
|
|
33
|
-
* →
|
|
34
|
-
* (
|
|
35
|
-
*
|
|
35
|
+
* → agent_before_settle actionable
|
|
36
|
+
* boundary (0.87.0):
|
|
37
|
+
* { entries: [continuation draft],
|
|
38
|
+
* continue: true } — exactly one
|
|
39
|
+
* next provider request; sendMessage
|
|
40
|
+
* fallback kept for continuations
|
|
41
|
+
* withheld after settlement
|
|
36
42
|
* setActiveToolsByName → pi.setActiveTools
|
|
37
43
|
* status-line segment → ctx.ui.setStatus("goal", ...)
|
|
38
44
|
* settings goal.enabled / continuationModes / statusInFooter
|
|
@@ -59,7 +65,12 @@
|
|
|
59
65
|
import { readFileSync } from "node:fs";
|
|
60
66
|
import * as path from "node:path";
|
|
61
67
|
import { fileURLToPath } from "node:url";
|
|
62
|
-
import type {
|
|
68
|
+
import type {
|
|
69
|
+
ContextEditEntryDraft,
|
|
70
|
+
ExtensionAPI,
|
|
71
|
+
ExtensionContext,
|
|
72
|
+
SessionBoundaryDraft,
|
|
73
|
+
} from "@earendil-works/pi-coding-agent";
|
|
63
74
|
import { Text } from "@earendil-works/pi-tui";
|
|
64
75
|
import {
|
|
65
76
|
createGoalCommand,
|
|
@@ -97,6 +108,7 @@ const goalModeContextPrompt = readFileSync(
|
|
|
97
108
|
interface EntryMessageLike {
|
|
98
109
|
role?: string;
|
|
99
110
|
stopReason?: string;
|
|
111
|
+
errorMessage?: string;
|
|
100
112
|
usage?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined;
|
|
101
113
|
}
|
|
102
114
|
|
|
@@ -104,6 +116,34 @@ interface EntryUsageLike {
|
|
|
104
116
|
usage?: EntryMessageLike["usage"];
|
|
105
117
|
}
|
|
106
118
|
|
|
119
|
+
/** Budget slimming drafts (opt-in via PI_GOAL_SLIM_ON_BUDGET=1, attached once
|
|
120
|
+
* per goal at the budget-limited flip): omit toolResult entries that precede
|
|
121
|
+
* the newest real user message from future provider context via append-only
|
|
122
|
+
* context-edit drafts — superseded turns' work products. Raw history, usage
|
|
123
|
+
* and UI history stay untouched (context_edit contract). Pure: returns the
|
|
124
|
+
* drafts; the settle boundary attaches them. */
|
|
125
|
+
export function contextSlimDrafts(branch: unknown[]): ContextEditEntryDraft[] {
|
|
126
|
+
const drafts: ContextEditEntryDraft[] = [];
|
|
127
|
+
let keepFrom = -1;
|
|
128
|
+
for (let i = branch.length - 1; i >= 0; i--) {
|
|
129
|
+
const entry = branch[i] as { type?: string; message?: { role?: string } } | undefined;
|
|
130
|
+
if (entry?.type === "message" && entry.message?.role === "user") {
|
|
131
|
+
keepFrom = i;
|
|
132
|
+
break;
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
if (keepFrom <= 0) return drafts;
|
|
136
|
+
const seen = new Set<string>();
|
|
137
|
+
for (let i = 0; i < keepFrom; i++) {
|
|
138
|
+
const entry = branch[i] as { type?: string; id?: string; message?: { role?: string } } | undefined;
|
|
139
|
+
if (entry?.type !== "message" || entry.message?.role !== "toolResult") continue;
|
|
140
|
+
if (!entry.id || seen.has(entry.id)) continue;
|
|
141
|
+
seen.add(entry.id);
|
|
142
|
+
drafts.push({ type: "context_edit", targetId: entry.id, replacement: null });
|
|
143
|
+
}
|
|
144
|
+
return drafts;
|
|
145
|
+
}
|
|
146
|
+
|
|
107
147
|
export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
108
148
|
// ------------------------------------------------------------------
|
|
109
149
|
// Closure state (omp session fields)
|
|
@@ -125,6 +165,10 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
125
165
|
/** A blocking ctx.ui dialog is open (ui_prompt_start/end). While open, no
|
|
126
166
|
* continuation may fire: the modal would hide a turn starting behind it. */
|
|
127
167
|
let uiPromptOpen = false;
|
|
168
|
+
/** Goal id whose budget-limited flip already armed context slimming. */
|
|
169
|
+
let slimmedBudgetGoalId: string | undefined;
|
|
170
|
+
/** Context-slimming drafts pending attachment at the next settle boundary. */
|
|
171
|
+
let pendingSlim = false;
|
|
128
172
|
|
|
129
173
|
// ------------------------------------------------------------------
|
|
130
174
|
// Usage accounting (mirror of pi getSessionStats().tokens sums)
|
|
@@ -256,6 +300,14 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
256
300
|
return;
|
|
257
301
|
}
|
|
258
302
|
goalState = state;
|
|
303
|
+
if (
|
|
304
|
+
state?.goal.status === "budget-limited" &&
|
|
305
|
+
process.env.PI_GOAL_SLIM_ON_BUDGET === "1" &&
|
|
306
|
+
slimmedBudgetGoalId !== state.goal.id
|
|
307
|
+
) {
|
|
308
|
+
slimmedBudgetGoalId = state.goal.id;
|
|
309
|
+
pendingSlim = true;
|
|
310
|
+
}
|
|
259
311
|
if (!state?.enabled) {
|
|
260
312
|
continuationInFlight = false;
|
|
261
313
|
}
|
|
@@ -340,6 +392,23 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
340
392
|
await exitGoalMode({ reason: "paused" });
|
|
341
393
|
}
|
|
342
394
|
|
|
395
|
+
/** A provider/agent error ended the run (borrowed from mitsuhiko/agent-stuff
|
|
396
|
+
* goal.ts): pause the active goal instead of letting scheduleContinuation
|
|
397
|
+
* fire into a likely retry loop, and classify the failure for the user.
|
|
398
|
+
* Usage/rate/quota errors read differently from generic faults. */
|
|
399
|
+
async function pauseOnRunError(messages: unknown[]): Promise<void> {
|
|
400
|
+
if (!goalState?.enabled || goalState.goal.status !== "active") return;
|
|
401
|
+
const errorMessage = lastAssistantErrorMessage(messages) ?? "";
|
|
402
|
+
const usageLimited = /\b(usage|rate|quota|limit)\b/i.test(errorMessage);
|
|
403
|
+
await runtime.pauseGoal();
|
|
404
|
+
notify(
|
|
405
|
+
usageLimited
|
|
406
|
+
? "Goal paused: the last turn hit provider usage/rate limits. /goal resume when ready."
|
|
407
|
+
: "Goal paused: the last turn ended with an error. /goal resume to continue.",
|
|
408
|
+
"error",
|
|
409
|
+
);
|
|
410
|
+
}
|
|
411
|
+
|
|
343
412
|
async function dropGoal(): Promise<void> {
|
|
344
413
|
if (!goalState) {
|
|
345
414
|
notify("No goal to drop.", "warning");
|
|
@@ -398,7 +467,9 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
398
467
|
return renderTemplate(goalModeContextPrompt, { goalContext: content, todoContext });
|
|
399
468
|
}
|
|
400
469
|
|
|
401
|
-
/** omp #scheduleGoalContinuation
|
|
470
|
+
/** omp #scheduleGoalContinuation. Post-settlement fallback path only: the
|
|
471
|
+
* primary continuation decision lives in the agent_before_settle boundary
|
|
472
|
+
* below (0.87.0). */
|
|
402
473
|
function scheduleContinuation(): void {
|
|
403
474
|
if (!goalState?.enabled || goalState.goal.status !== "active") return;
|
|
404
475
|
if (suppressNextContinuation) return;
|
|
@@ -408,11 +479,33 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
408
479
|
if (!prompt) return;
|
|
409
480
|
continuationInFlight = true;
|
|
410
481
|
pi.sendMessage(
|
|
411
|
-
|
|
482
|
+
// details.goalId keys context pruning (see the "context" handler below):
|
|
483
|
+
// only the newest continuation of the CURRENTLY active goal survives.
|
|
484
|
+
{ customType: "goal-continuation", content: prompt, display: false, details: { goalId: goalState.goal.id } },
|
|
412
485
|
{ triggerTurn: true, deliverAs: "followUp" },
|
|
413
486
|
);
|
|
414
487
|
}
|
|
415
488
|
|
|
489
|
+
/** omp #scheduleGoalContinuation decision chain, settled into a draft:
|
|
490
|
+
* guards pass → one custom_message draft carrying the continuation
|
|
491
|
+
* prompt (details.goalId keys context pruning). */
|
|
492
|
+
function continuationDraft(ctx: ExtensionContext): SessionBoundaryDraft | undefined {
|
|
493
|
+
if (!goalState?.enabled || goalState.goal.status !== "active") return undefined;
|
|
494
|
+
if (suppressNextContinuation) return undefined;
|
|
495
|
+
if (uiPromptOpen) return undefined; // modal open: never start a turn behind it
|
|
496
|
+
if (ctx.hasPendingMessages()) return undefined;
|
|
497
|
+
const prompt = runtime.buildContinuationPrompt();
|
|
498
|
+
if (!prompt) return undefined;
|
|
499
|
+
continuationInFlight = true;
|
|
500
|
+
return {
|
|
501
|
+
type: "custom_message",
|
|
502
|
+
customType: "goal-continuation",
|
|
503
|
+
content: prompt,
|
|
504
|
+
display: false,
|
|
505
|
+
details: { goalId: goalState.goal.id },
|
|
506
|
+
};
|
|
507
|
+
}
|
|
508
|
+
|
|
416
509
|
// ------------------------------------------------------------------
|
|
417
510
|
// Tool + command registration
|
|
418
511
|
// ------------------------------------------------------------------
|
|
@@ -425,6 +518,11 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
425
518
|
// fall back to the process cwd.
|
|
426
519
|
runEvaluator: (request, opts) =>
|
|
427
520
|
runGoalEvaluator(request, { cwd: opts.cwd ?? currentCtx?.cwd ?? process.cwd(), signal: opts.signal }),
|
|
521
|
+
// Mid-run activation (tool create/resume): inject the goal context into
|
|
522
|
+
// the CURRENT run as a steer — before_agent_start only covers the next run.
|
|
523
|
+
onActivated: async () => {
|
|
524
|
+
if (currentCtx && !currentCtx.isIdle()) await sendGoalModeContext("steer");
|
|
525
|
+
},
|
|
428
526
|
};
|
|
429
527
|
pi.registerTool(createGoalTool(goalToolDeps));
|
|
430
528
|
|
|
@@ -501,11 +599,16 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
501
599
|
const restored = restoreGoalFromEntries(branch);
|
|
502
600
|
if (!restored) {
|
|
503
601
|
goalState = undefined;
|
|
504
|
-
// omp sdk.ts excludes the goal tool from the initial set; mirror that
|
|
505
|
-
//
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
602
|
+
// omp sdk.ts excludes the goal tool from the initial set; mirror that in
|
|
603
|
+
// TUI where /goal and /guided-goal re-arm it. In non-interactive modes
|
|
604
|
+
// (print/json/rpc) no slash command exists, so removing the tool would
|
|
605
|
+
// leave the model unable to create a goal at all (found via live
|
|
606
|
+
// goal-mode testing): keep it active there.
|
|
607
|
+
if (ctx.mode === "tui") {
|
|
608
|
+
const active = pi.getActiveTools();
|
|
609
|
+
if (active.includes("goal")) {
|
|
610
|
+
pi.setActiveTools(active.filter((name) => name !== "goal"));
|
|
611
|
+
}
|
|
509
612
|
}
|
|
510
613
|
updateStatus();
|
|
511
614
|
return;
|
|
@@ -554,12 +657,59 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
554
657
|
}
|
|
555
658
|
});
|
|
556
659
|
|
|
557
|
-
|
|
660
|
+
// Goal context injection (0.87.0 context_with_system): runs after `context`
|
|
661
|
+
// handlers on the full transcript including system messages; the result is
|
|
662
|
+
// sent verbatim. Per-request injection survives compaction (re-applied every
|
|
663
|
+
// call) and the fresh copy supersedes any steer-injected transcript copies.
|
|
664
|
+
pi.on("context_with_system", (event, ctx) => {
|
|
558
665
|
currentCtx = ctx;
|
|
559
666
|
const content = buildGoalModeMessage();
|
|
560
667
|
if (!content) return undefined;
|
|
668
|
+
const messages = event.messages.filter((message) => {
|
|
669
|
+
const msg = message as { role?: string; customType?: string } | undefined;
|
|
670
|
+
return !(msg?.role === "custom" && msg.customType === "goal-mode-context");
|
|
671
|
+
});
|
|
672
|
+
messages.push({
|
|
673
|
+
role: "custom",
|
|
674
|
+
customType: "goal-mode-context",
|
|
675
|
+
content,
|
|
676
|
+
display: false,
|
|
677
|
+
} as (typeof event.messages)[number]);
|
|
678
|
+
return { messages };
|
|
679
|
+
});
|
|
680
|
+
|
|
681
|
+
// Context hygiene (borrowed from mitsuhiko/agent-stuff goal.ts): hidden goal
|
|
682
|
+
// messages otherwise accumulate one per run/continuation and burn context
|
|
683
|
+
// every LLM call. Keep only the newest goal-mode-context and
|
|
684
|
+
// goal-budget-limit, plus the newest continuation stamped for the currently
|
|
685
|
+
// active goal id; stale ones (including all continuations once no goal is
|
|
686
|
+
// active) are dropped from the model's view, not from the transcript.
|
|
687
|
+
pi.on("context", (event) => {
|
|
688
|
+
const activeGoalId = goalState?.enabled && goalState.goal.status === "active" ? goalState.goal.id : undefined;
|
|
689
|
+
let lastContext = -1;
|
|
690
|
+
let lastBudget = -1;
|
|
691
|
+
let lastContinuation = -1;
|
|
692
|
+
for (let i = 0; i < event.messages.length; i++) {
|
|
693
|
+
const msg = event.messages[i] as
|
|
694
|
+
| { role?: string; customType?: string; details?: { goalId?: string } }
|
|
695
|
+
| undefined;
|
|
696
|
+
if (msg?.role !== "custom") continue;
|
|
697
|
+
if (msg.customType === "goal-mode-context") lastContext = i;
|
|
698
|
+
else if (msg.customType === "goal-budget-limit") lastBudget = i;
|
|
699
|
+
else if (msg.customType === "goal-continuation" && activeGoalId !== undefined && msg.details?.goalId === activeGoalId) {
|
|
700
|
+
lastContinuation = i;
|
|
701
|
+
}
|
|
702
|
+
}
|
|
703
|
+
if (lastContext === -1 && lastBudget === -1 && lastContinuation === -1) return undefined;
|
|
561
704
|
return {
|
|
562
|
-
|
|
705
|
+
messages: event.messages.filter((message, index) => {
|
|
706
|
+
const msg = message as { role?: string; customType?: string } | undefined;
|
|
707
|
+
if (msg?.role !== "custom") return true;
|
|
708
|
+
if (msg.customType === "goal-mode-context") return index === lastContext;
|
|
709
|
+
if (msg.customType === "goal-budget-limit") return index === lastBudget;
|
|
710
|
+
if (msg.customType === "goal-continuation") return index === lastContinuation;
|
|
711
|
+
return true;
|
|
712
|
+
}),
|
|
563
713
|
};
|
|
564
714
|
});
|
|
565
715
|
|
|
@@ -567,11 +717,12 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
567
717
|
currentCtx = ctx;
|
|
568
718
|
// omp separates onAgentEnd (session) from continuation scheduling
|
|
569
719
|
// (interactive-mode); both subscribe to the same end-of-run moment.
|
|
570
|
-
const
|
|
571
|
-
if (aborted) {
|
|
720
|
+
const stopReason = lastAssistantStopReason(event.messages);
|
|
721
|
+
if (stopReason === "aborted") {
|
|
572
722
|
await runtime.onTaskAborted({ reason: "interrupted" });
|
|
573
723
|
} else {
|
|
574
724
|
await runtime.onAgentEnd({ currentUsage: currentUsage(ctx) });
|
|
725
|
+
if (stopReason === "error") await pauseOnRunError(event.messages);
|
|
575
726
|
}
|
|
576
727
|
|
|
577
728
|
if (continuationInFlight) {
|
|
@@ -583,7 +734,31 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
|
|
|
583
734
|
return;
|
|
584
735
|
}
|
|
585
736
|
updateStatus();
|
|
586
|
-
|
|
737
|
+
});
|
|
738
|
+
|
|
739
|
+
// omp #scheduleGoalContinuation migrated to the 0.87.0 actionable boundary:
|
|
740
|
+
// at settle time decide ONCE whether exactly one more provider request
|
|
741
|
+
// should run, attaching the continuation prompt as a custom_message draft.
|
|
742
|
+
// No timer, no followUp scheduling — pi owns the request. Guards mirror the
|
|
743
|
+
// old scheduleContinuation chain; continuations withheld after settlement
|
|
744
|
+
// (modal closed late, budget set while idle) still recover via the
|
|
745
|
+
// sendMessage fallback in scheduleContinuation().
|
|
746
|
+
pi.on("agent_before_settle", (_event, ctx) => {
|
|
747
|
+
currentCtx = ctx;
|
|
748
|
+
// Context slimming (opt-in): attach pending context_edit drafts first —
|
|
749
|
+
// they persist even when the continuation itself is withheld.
|
|
750
|
+
const drafts: SessionBoundaryDraft[] = [];
|
|
751
|
+
if (pendingSlim) {
|
|
752
|
+
pendingSlim = false;
|
|
753
|
+
drafts.push(...contextSlimDrafts(ctx.sessionManager.getBranch()));
|
|
754
|
+
}
|
|
755
|
+
const continuation = continuationDraft(ctx);
|
|
756
|
+
if (continuation) drafts.push(continuation);
|
|
757
|
+
if (drafts.length === 0) return undefined;
|
|
758
|
+
// Absent `continue` = no opinion on auto-continuation (budget-limited
|
|
759
|
+
// goals settle without one); `continue: true` guarantees exactly one
|
|
760
|
+
// next provider request.
|
|
761
|
+
return continuation !== undefined ? { entries: drafts, continue: true } : { entries: drafts };
|
|
587
762
|
});
|
|
588
763
|
|
|
589
764
|
pi.on("turn_end", (_event, ctx) => {
|
|
@@ -665,6 +840,14 @@ function lastAssistantStopReason(messages: unknown[]): string | undefined {
|
|
|
665
840
|
return undefined;
|
|
666
841
|
}
|
|
667
842
|
|
|
843
|
+
function lastAssistantErrorMessage(messages: unknown[]): string | undefined {
|
|
844
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
845
|
+
const message = messages[i] as EntryMessageLike | undefined;
|
|
846
|
+
if (message?.role === "assistant") return message.errorMessage;
|
|
847
|
+
}
|
|
848
|
+
return undefined;
|
|
849
|
+
}
|
|
850
|
+
|
|
668
851
|
export type { GoalRuntimeHost } from "./src/runtime.ts";
|
|
669
852
|
// Re-exported for consumers that compose the pieces directly (tests, tools).
|
|
670
853
|
export { GoalRuntime } from "./src/runtime.ts";
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@fyeeme/pi-goal",
|
|
3
|
-
"version": "1.0.
|
|
4
|
-
"description": "Goal mode for pi
|
|
3
|
+
"version": "1.0.4",
|
|
4
|
+
"description": "Goal mode for pi — one persistent autonomous objective looped until verified success: goal tool (create/get/complete/resume/drop), token/time budget accounting with budget-limit steering, automatic continuation turns, /goal and /guided-goal commands, and a goal_updated event bus for other extensions.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
7
7
|
"author": "fyeeme",
|
|
@@ -47,17 +47,17 @@
|
|
|
47
47
|
"typecheck": "tsc"
|
|
48
48
|
},
|
|
49
49
|
"peerDependencies": {
|
|
50
|
-
"@earendil-works/pi-ai": ">=0.
|
|
51
|
-
"@earendil-works/pi-coding-agent": ">=0.
|
|
52
|
-
"@earendil-works/pi-tui": ">=0.
|
|
50
|
+
"@earendil-works/pi-ai": ">=0.99.0",
|
|
51
|
+
"@earendil-works/pi-coding-agent": ">=0.99.0",
|
|
52
|
+
"@earendil-works/pi-tui": ">=0.99.0",
|
|
53
53
|
"typebox": ">=1.0.0",
|
|
54
|
-
"@earendil-works/pi-agent-core": "0.
|
|
54
|
+
"@earendil-works/pi-agent-core": "0.99.2"
|
|
55
55
|
},
|
|
56
56
|
"devDependencies": {
|
|
57
|
-
"@earendil-works/pi-agent-core": "0.
|
|
58
|
-
"@earendil-works/pi-ai": "0.
|
|
59
|
-
"@earendil-works/pi-coding-agent": "0.
|
|
60
|
-
"@earendil-works/pi-tui": "0.
|
|
57
|
+
"@earendil-works/pi-agent-core": "0.99.2",
|
|
58
|
+
"@earendil-works/pi-ai": "0.99.2",
|
|
59
|
+
"@earendil-works/pi-coding-agent": "0.99.2",
|
|
60
|
+
"@earendil-works/pi-tui": "0.99.2",
|
|
61
61
|
"@types/node": "22.19.19",
|
|
62
62
|
"typescript": "5.9.3",
|
|
63
63
|
"vitest": "3.2.7"
|
package/src/evaluator.ts
CHANGED
|
@@ -34,8 +34,16 @@ const evaluatorCompletePrompt = readFileSync(path.join(promptsDir, "evaluator-co
|
|
|
34
34
|
const evaluatorImpossiblePrompt = readFileSync(path.join(promptsDir, "evaluator-impossible.md"), "utf8");
|
|
35
35
|
|
|
36
36
|
/** Hard cap on one evaluator run — the grounded check may run test suites,
|
|
37
|
-
* so this is deliberately generous; anything longer has failed.
|
|
38
|
-
|
|
37
|
+
* so this is deliberately generous; anything longer has failed. Each LLM
|
|
38
|
+
* call on slow providers can take 20s+, and the evaluator is a multi-step
|
|
39
|
+
* agent (inspect repo, run checks, emit JSON), so the default needs real
|
|
40
|
+
* headroom. Override with GOAL_EVALUATOR_TIMEOUT_MS. */
|
|
41
|
+
export const EVALUATOR_TIMEOUT_MS = 600_000;
|
|
42
|
+
|
|
43
|
+
function evaluatorTimeout(): number {
|
|
44
|
+
const raw = Number.parseInt(process.env.GOAL_EVALUATOR_TIMEOUT_MS ?? "", 10);
|
|
45
|
+
return Number.isInteger(raw) && raw > 0 ? raw : EVALUATOR_TIMEOUT_MS;
|
|
46
|
+
}
|
|
39
47
|
|
|
40
48
|
/** Argv-safety caps (kernel MAX_ARG_STRLEN ≈ 128KB; these keep the prompt
|
|
41
49
|
* argument far under it even for verbose objectives/audits). */
|
|
@@ -118,6 +126,16 @@ export function buildEvaluatorPrompt(request: GoalEvaluatorRequest): string {
|
|
|
118
126
|
});
|
|
119
127
|
}
|
|
120
128
|
|
|
129
|
+
/** Build the evaluator argv: a one-shot headless run. Lean flags matter here:
|
|
130
|
+
* without them the nested pi loads the user's full extension set — including
|
|
131
|
+
* any OTHER goal extension, whose active-goal system prompt would pollute the
|
|
132
|
+
* very evaluator that is supposed to judge the goal independently — and its
|
|
133
|
+
* startup cost pushes long verification runs over the timeout (observed
|
|
134
|
+
* live: 300s timeout hit while the evaluator was still working). */
|
|
135
|
+
function evaluatorInvocation(prompt: string): { command: string; args: string[] } {
|
|
136
|
+
return getPiInvocation(["-p", "--no-session", "--no-extensions", "--no-skills", prompt]);
|
|
137
|
+
}
|
|
138
|
+
|
|
121
139
|
/** Extract the first balanced JSON object from evaluator output. Handles the
|
|
122
140
|
* clean case, code-fenced output, and prose-wrapped JSON; anything else is
|
|
123
141
|
* unparseable. Never throws. */
|
|
@@ -181,10 +199,10 @@ export async function runGoalEvaluator(
|
|
|
181
199
|
): Promise<GoalEvaluatorOutcome> {
|
|
182
200
|
const prompt = buildEvaluatorPrompt(request);
|
|
183
201
|
const spawn = opts.spawn ?? defaultSpawn;
|
|
184
|
-
const timeoutMs = opts.timeoutMs ??
|
|
202
|
+
const timeoutMs = opts.timeoutMs ?? evaluatorTimeout();
|
|
185
203
|
let stdout: string;
|
|
186
204
|
try {
|
|
187
|
-
stdout = await spawn(
|
|
205
|
+
stdout = await spawn(evaluatorInvocation(prompt), {
|
|
188
206
|
cwd: opts.cwd,
|
|
189
207
|
timeoutMs,
|
|
190
208
|
signal: opts.signal,
|
package/src/runtime.ts
CHANGED
|
@@ -128,6 +128,25 @@ export function completionBudgetReport(goal: Goal): string | null {
|
|
|
128
128
|
return `Goal achieved. Report final budget usage to the user: ${parts.join("; ")}.`;
|
|
129
129
|
}
|
|
130
130
|
|
|
131
|
+
/** Objective length cap. The objective is re-injected into context on every
|
|
132
|
+
* continuation and passed to the evaluator subprocess, so an oversized one
|
|
133
|
+
* silently burns budget every turn. Borrowed from mitsuhiko/agent-stuff
|
|
134
|
+
* goal.ts (MAX_OBJECTIVE_CHARS). */
|
|
135
|
+
export const MAX_OBJECTIVE_CHARS = 4_000;
|
|
136
|
+
|
|
137
|
+
function charCount(value: string): number {
|
|
138
|
+
return [...value].length;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function validateObjective(objective: string): void {
|
|
142
|
+
const count = charCount(objective);
|
|
143
|
+
if (count > MAX_OBJECTIVE_CHARS) {
|
|
144
|
+
throw new Error(
|
|
145
|
+
`Goal objective is too long: ${count.toLocaleString()} characters. Limit: ${MAX_OBJECTIVE_CHARS.toLocaleString()} characters. Put longer instructions in a file and reference that file in the objective.`,
|
|
146
|
+
);
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
131
150
|
function validateTokenBudget(tokenBudget: number | undefined): void {
|
|
132
151
|
if (tokenBudget !== undefined && (!Number.isInteger(tokenBudget) || tokenBudget <= 0)) {
|
|
133
152
|
throw new Error("goal token_budget must be a positive integer when provided");
|
|
@@ -408,6 +427,7 @@ export class GoalRuntime {
|
|
|
408
427
|
async createGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
|
|
409
428
|
const objective = input.objective.trim();
|
|
410
429
|
if (!objective) throw new Error("objective is required when op=create");
|
|
430
|
+
validateObjective(objective);
|
|
411
431
|
validateTokenBudget(input.tokenBudget);
|
|
412
432
|
return await this.#withAccounting(async () => {
|
|
413
433
|
const existing = this.#host.getState();
|
|
@@ -425,6 +445,7 @@ export class GoalRuntime {
|
|
|
425
445
|
async replaceGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
|
|
426
446
|
const objective = input.objective.trim();
|
|
427
447
|
if (!objective) throw new Error("objective is required when op=replace");
|
|
448
|
+
validateObjective(objective);
|
|
428
449
|
validateTokenBudget(input.tokenBudget);
|
|
429
450
|
return await this.#withAccounting(async () => {
|
|
430
451
|
const existing = this.#host.getState();
|
package/src/tool.ts
CHANGED
|
@@ -127,6 +127,11 @@ export interface GoalToolDeps {
|
|
|
127
127
|
/** Independent completion/impossibility evaluator (src/evaluator.ts).
|
|
128
128
|
* Gates `complete` and adjudicates `impossible`. */
|
|
129
129
|
runEvaluator: GoalEvaluatorFn;
|
|
130
|
+
/** Called after the tool ACTIVATES a goal mid-run (create/resume). The
|
|
131
|
+
* before_agent_start injection only fires on the next agent run, so
|
|
132
|
+
* without this hook the run that created the goal never sees the goal
|
|
133
|
+
* context prompt at all (observed live in print mode). */
|
|
134
|
+
onActivated?: () => Promise<void>;
|
|
130
135
|
}
|
|
131
136
|
|
|
132
137
|
function describeEvaluatorVerdict(verdict: string): string {
|
|
@@ -172,12 +177,14 @@ export function createGoalTool(deps: GoalToolDeps): ToolDefinition<typeof GoalPa
|
|
|
172
177
|
if (params.op === "create") {
|
|
173
178
|
const created = await runtime.createGoal(validateCreateParams(params));
|
|
174
179
|
response = buildGoalToolResponse(created.goal);
|
|
180
|
+
await deps.onActivated?.();
|
|
175
181
|
} else if (params.op === "get") {
|
|
176
182
|
const state = deps.getState();
|
|
177
183
|
response = buildGoalToolResponse(state?.goal ?? null);
|
|
178
184
|
} else if (params.op === "resume") {
|
|
179
185
|
const resumed = await runtime.resumeGoal();
|
|
180
186
|
response = buildGoalToolResponse(resumed.goal);
|
|
187
|
+
await deps.onActivated?.();
|
|
181
188
|
} else if (params.op === "drop") {
|
|
182
189
|
const dropped = await runtime.dropGoal();
|
|
183
190
|
response = buildGoalToolResponse(dropped ?? null);
|
package/test/evaluator.test.ts
CHANGED
|
@@ -15,7 +15,6 @@ import {
|
|
|
15
15
|
runGoalEvaluator,
|
|
16
16
|
type EvaluatorSpawn,
|
|
17
17
|
} from "../src/evaluator.ts";
|
|
18
|
-
|
|
19
18
|
function spawnReturning(stdout: string): { spawn: EvaluatorSpawn; calls: { args: string[]; cwd: string }[] } {
|
|
20
19
|
const calls: { args: string[]; cwd: string }[] = [];
|
|
21
20
|
return {
|
|
@@ -31,6 +30,38 @@ const COMPLETE_REQUEST = { mode: "complete" as const, objective: "Ship it", clai
|
|
|
31
30
|
const IMPOSSIBLE_REQUEST = { mode: "impossible" as const, objective: "Ship it", claim: "no network access" };
|
|
32
31
|
const RUN_OPTS = { cwd: "/tmp/repo" };
|
|
33
32
|
|
|
33
|
+
describe("evaluator subprocess invocation", () => {
|
|
34
|
+
it("runs lean: no extensions, no skills (a nested goal extension would pollute the judge)", async () => {
|
|
35
|
+
const run = spawnReturning('{"ok": true}');
|
|
36
|
+
await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: run.spawn });
|
|
37
|
+
const { args } = run.calls[0]!;
|
|
38
|
+
expect(args).toContain("--no-extensions");
|
|
39
|
+
expect(args).toContain("--no-skills");
|
|
40
|
+
// Prompt stays last (argv-safety caps rely on it).
|
|
41
|
+
expect(args.indexOf("--no-skills")).toBeLessThan(args.length - 1);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
it("honors GOAL_EVALUATOR_TIMEOUT_MS over the default", async () => {
|
|
45
|
+
const run = spawnReturning('{"ok": true}');
|
|
46
|
+
const timeouts: number[] = [];
|
|
47
|
+
const previous = process.env.GOAL_EVALUATOR_TIMEOUT_MS;
|
|
48
|
+
process.env.GOAL_EVALUATOR_TIMEOUT_MS = "12345";
|
|
49
|
+
try {
|
|
50
|
+
await runGoalEvaluator(COMPLETE_REQUEST, {
|
|
51
|
+
...RUN_OPTS,
|
|
52
|
+
spawn: async (invocation, opts) => {
|
|
53
|
+
timeouts.push(opts.timeoutMs);
|
|
54
|
+
return run.spawn(invocation, opts);
|
|
55
|
+
},
|
|
56
|
+
});
|
|
57
|
+
} finally {
|
|
58
|
+
if (previous === undefined) delete process.env.GOAL_EVALUATOR_TIMEOUT_MS;
|
|
59
|
+
else process.env.GOAL_EVALUATOR_TIMEOUT_MS = previous;
|
|
60
|
+
}
|
|
61
|
+
expect(timeouts).toEqual([12_345]);
|
|
62
|
+
});
|
|
63
|
+
});
|
|
64
|
+
|
|
34
65
|
describe("extractJsonObject", () => {
|
|
35
66
|
it("parses a clean JSON object", () => {
|
|
36
67
|
expect(extractJsonObject('{"ok": true, "reason": "tests pass"}')).toEqual({
|
package/test/index.test.ts
CHANGED
|
@@ -36,6 +36,7 @@ interface FakeHost {
|
|
|
36
36
|
customType: string;
|
|
37
37
|
content: string;
|
|
38
38
|
display: boolean;
|
|
39
|
+
details?: unknown;
|
|
39
40
|
options?: { deliverAs?: string; triggerTurn?: boolean };
|
|
40
41
|
}>;
|
|
41
42
|
sentUserMessages: Array<{ content: string; options?: { deliverAs?: string } }>;
|
|
@@ -114,7 +115,10 @@ function fakeHost(): FakeHost {
|
|
|
114
115
|
return host;
|
|
115
116
|
}
|
|
116
117
|
|
|
117
|
-
function createContext(
|
|
118
|
+
function createContext(
|
|
119
|
+
host: FakeHost,
|
|
120
|
+
overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean; mode?: string } = {},
|
|
121
|
+
) {
|
|
118
122
|
return {
|
|
119
123
|
ui: {
|
|
120
124
|
notify: (text: string) => {
|
|
@@ -130,7 +134,7 @@ function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending
|
|
|
130
134
|
editor: async () => undefined,
|
|
131
135
|
},
|
|
132
136
|
hasUI: true,
|
|
133
|
-
mode: "tui",
|
|
137
|
+
mode: overrides.mode ?? "tui",
|
|
134
138
|
cwd: "/tmp",
|
|
135
139
|
isIdle: () => overrides.idle ?? true,
|
|
136
140
|
hasPendingMessages: () => overrides.pending ?? false,
|
|
@@ -148,6 +152,19 @@ async function fire(host: FakeHost, event: string, payload: unknown, ctx?: unkno
|
|
|
148
152
|
}
|
|
149
153
|
}
|
|
150
154
|
|
|
155
|
+
/** Fire agent_before_settle and return its boundary result. */
|
|
156
|
+
function fireBeforeSettle(
|
|
157
|
+
host: FakeHost,
|
|
158
|
+
ctx?: unknown,
|
|
159
|
+
): { continue?: boolean; entries?: Array<{ type: string; customType?: string; details?: unknown }> } | undefined {
|
|
160
|
+
const list = host.handlers.get("agent_before_settle") ?? [];
|
|
161
|
+
let result: unknown;
|
|
162
|
+
for (const handler of list) {
|
|
163
|
+
result = handler({ type: "agent_before_settle" }, ctx ?? createContext(host));
|
|
164
|
+
}
|
|
165
|
+
return result as { continue?: boolean; entries?: Array<{ type: string; customType?: string; details?: unknown }> } | undefined;
|
|
166
|
+
}
|
|
167
|
+
|
|
151
168
|
function fireBus(host: FakeHost, channel: string, data: unknown): void {
|
|
152
169
|
for (const handler of host.busHandlers.get(channel) ?? []) {
|
|
153
170
|
handler(data);
|
|
@@ -222,6 +239,14 @@ describe("pi-goal extension wiring", () => {
|
|
|
222
239
|
expect(host.activeTools).toContain("read");
|
|
223
240
|
});
|
|
224
241
|
|
|
242
|
+
it("session_start keeps the goal tool in non-interactive modes (no slash commands there)", async () => {
|
|
243
|
+
const host = fakeHost();
|
|
244
|
+
defaultExport(host.pi);
|
|
245
|
+
const ctx = createContext(host, { mode: "print" });
|
|
246
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" }, ctx);
|
|
247
|
+
expect(host.activeTools).toContain("goal");
|
|
248
|
+
});
|
|
249
|
+
|
|
225
250
|
it("session_start restores a persisted active goal, re-adds the tool, then pauses it (omp onThreadResumed)", async () => {
|
|
226
251
|
const host = fakeHost();
|
|
227
252
|
defaultExport(host.pi);
|
|
@@ -256,30 +281,52 @@ describe("pi-goal extension wiring", () => {
|
|
|
256
281
|
expect(host.statuses.get("goal")).toBe("🎯 Goal 0");
|
|
257
282
|
});
|
|
258
283
|
|
|
259
|
-
it("
|
|
284
|
+
it("context_with_system injects the goal-mode-context message only while a goal is active", async () => {
|
|
260
285
|
const host = fakeHost();
|
|
261
286
|
defaultExport(host.pi);
|
|
262
287
|
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
263
288
|
|
|
264
|
-
|
|
265
|
-
const
|
|
266
|
-
result = await handler
|
|
289
|
+
const handler = host.handlers.get("context_with_system")?.[0]!;
|
|
290
|
+
const transcript = [{ role: "user", content: "hi" }];
|
|
291
|
+
let result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as
|
|
292
|
+
| { messages: Array<{ customType?: string; content?: string }> }
|
|
293
|
+
| undefined;
|
|
267
294
|
expect(result).toBeUndefined();
|
|
268
295
|
|
|
269
296
|
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
270
|
-
result = await handler
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
expect(
|
|
297
|
+
result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as
|
|
298
|
+
| { messages: Array<{ customType?: string; content?: string }> }
|
|
299
|
+
| undefined;
|
|
300
|
+
const contexts = result!.messages.filter((m) => m.customType === "goal-mode-context");
|
|
301
|
+
expect(contexts).toHaveLength(1);
|
|
302
|
+
expect(contexts[0]?.content).toContain("<goal_context>");
|
|
303
|
+
expect(contexts[0]?.content).toContain("Do the thing");
|
|
275
304
|
});
|
|
276
305
|
|
|
277
|
-
it("
|
|
306
|
+
it("context_with_system supersedes steer-injected transcript copies (single fresh context)", async () => {
|
|
307
|
+
const host = fakeHost();
|
|
308
|
+
defaultExport(host.pi);
|
|
309
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
310
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
311
|
+
|
|
312
|
+
const handler = host.handlers.get("context_with_system")?.[0]!;
|
|
313
|
+
const transcript = [
|
|
314
|
+
{ role: "user", content: "hi" },
|
|
315
|
+
{ role: "custom", customType: "goal-mode-context", content: "stale copy", display: false },
|
|
316
|
+
];
|
|
317
|
+
const result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as {
|
|
318
|
+
messages: Array<{ customType?: string; content?: string }>;
|
|
319
|
+
};
|
|
320
|
+
const contexts = result.messages.filter((m) => m.customType === "goal-mode-context");
|
|
321
|
+
expect(contexts).toHaveLength(1);
|
|
322
|
+
expect(contexts[0]?.content).not.toBe("stale copy");
|
|
323
|
+
});
|
|
324
|
+
|
|
325
|
+
it("agent_before_settle continues exactly one request when the goal is still active", async () => {
|
|
278
326
|
const host = fakeHost();
|
|
279
327
|
defaultExport(host.pi);
|
|
280
328
|
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
281
329
|
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
282
|
-
host.sentMessages.length = 0;
|
|
283
330
|
|
|
284
331
|
await fire(host, "agent_start", { type: "agent_start" });
|
|
285
332
|
await fire(host, "tool_execution_end", {
|
|
@@ -291,10 +338,16 @@ describe("pi-goal extension wiring", () => {
|
|
|
291
338
|
});
|
|
292
339
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("toolUse")] });
|
|
293
340
|
|
|
294
|
-
const
|
|
295
|
-
expect(
|
|
296
|
-
|
|
297
|
-
|
|
341
|
+
const result = fireBeforeSettle(host);
|
|
342
|
+
expect(result?.continue).toBe(true);
|
|
343
|
+
// The continuation prompt rides as a custom_message draft.
|
|
344
|
+
const draft = result?.entries?.[0];
|
|
345
|
+
expect(draft).toMatchObject({ type: "custom_message", customType: "goal-continuation", display: false });
|
|
346
|
+
// details.goalId keys the context-pruning handler.
|
|
347
|
+
const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
|
|
348
|
+
expect(draft?.details).toMatchObject({ goalId });
|
|
349
|
+
// The boundary path sends no followUp message.
|
|
350
|
+
expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
|
|
298
351
|
});
|
|
299
352
|
|
|
300
353
|
it("suppresses the next continuation when a continuation turn produced no tool calls", async () => {
|
|
@@ -302,18 +355,16 @@ describe("pi-goal extension wiring", () => {
|
|
|
302
355
|
defaultExport(host.pi);
|
|
303
356
|
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
304
357
|
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
305
|
-
host.sentMessages.length = 0;
|
|
306
358
|
|
|
307
359
|
// Continuation turn: agent replies with no tool calls.
|
|
308
360
|
await fire(host, "agent_start", { type: "agent_start" });
|
|
309
361
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
|
|
310
|
-
expect(host
|
|
362
|
+
expect(fireBeforeSettle(host)?.continue).toBe(true);
|
|
311
363
|
|
|
312
|
-
// That turn's end marks suppression: the following
|
|
313
|
-
host.sentMessages.length = 0;
|
|
364
|
+
// That turn's end marks suppression: the following settle continues nothing.
|
|
314
365
|
await fire(host, "agent_start", { type: "agent_start" });
|
|
315
366
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
|
|
316
|
-
expect(host
|
|
367
|
+
expect(fireBeforeSettle(host)?.continue).toBeUndefined();
|
|
317
368
|
|
|
318
369
|
// A real user message re-arms the loop.
|
|
319
370
|
await fire(host, "message_start", { type: "message_start", message: { role: "user" } });
|
|
@@ -325,9 +376,8 @@ describe("pi-goal extension wiring", () => {
|
|
|
325
376
|
result: {},
|
|
326
377
|
isError: false,
|
|
327
378
|
});
|
|
328
|
-
host.sentMessages.length = 0;
|
|
329
379
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
|
|
330
|
-
expect(host
|
|
380
|
+
expect(fireBeforeSettle(host)?.continue).toBe(true);
|
|
331
381
|
});
|
|
332
382
|
|
|
333
383
|
it("interrupt aborts pause the goal instead of continuing", async () => {
|
|
@@ -340,7 +390,7 @@ describe("pi-goal extension wiring", () => {
|
|
|
340
390
|
await fire(host, "agent_start", { type: "agent_start" });
|
|
341
391
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("aborted")] });
|
|
342
392
|
|
|
343
|
-
expect(host
|
|
393
|
+
expect(fireBeforeSettle(host)?.continue).toBeUndefined();
|
|
344
394
|
// omp footer segment: pause icon + usage.
|
|
345
395
|
expect(host.statuses.get("goal")).toBe("⏸ Goal 0");
|
|
346
396
|
const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
|
|
@@ -365,7 +415,7 @@ describe("pi-goal extension wiring", () => {
|
|
|
365
415
|
expect(host.entries.some((e) => e.customType === GOAL_CLEARED_ENTRY_TYPE)).toBe(true);
|
|
366
416
|
expect(host.notifications).toContain("Goal mode completed.");
|
|
367
417
|
expect(host.statuses.has("goal")).toBe(false);
|
|
368
|
-
expect(host
|
|
418
|
+
expect(fireBeforeSettle(host)?.continue).toBeUndefined();
|
|
369
419
|
});
|
|
370
420
|
|
|
371
421
|
it("budget-limit steering sends one hidden steer message when usage crosses the budget", async () => {
|
|
@@ -428,6 +478,96 @@ describe("pi-goal extension wiring", () => {
|
|
|
428
478
|
expect(host.sentMessages.filter((m) => m.customType === "goal-budget-limit")).toHaveLength(0);
|
|
429
479
|
});
|
|
430
480
|
|
|
481
|
+
it("budget flip attaches context-slimming drafts when PI_GOAL_SLIM_ON_BUDGET=1", async () => {
|
|
482
|
+
const host = fakeHost();
|
|
483
|
+
defaultExport(host.pi);
|
|
484
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
485
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
486
|
+
await host.commands.goal!.handler("budget 10", createContext(host));
|
|
487
|
+
|
|
488
|
+
const entries: unknown[] = [
|
|
489
|
+
{ type: "message", id: "t-old", message: { role: "toolResult" } },
|
|
490
|
+
{ type: "message", id: "u1", message: { role: "user" } },
|
|
491
|
+
{ type: "message", id: "t-new", message: { role: "toolResult" } },
|
|
492
|
+
];
|
|
493
|
+
const ctx = createContext(host, { entries });
|
|
494
|
+
|
|
495
|
+
vi.stubEnv("PI_GOAL_SLIM_ON_BUDGET", "1");
|
|
496
|
+
await fire(host, "turn_start", { type: "turn_start", turnIndex: 0, timestamp: 0 }, ctx);
|
|
497
|
+
entries.push({
|
|
498
|
+
type: "message",
|
|
499
|
+
id: "m1",
|
|
500
|
+
message: { role: "assistant", stopReason: "toolUse", usage: { input: 25, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
501
|
+
});
|
|
502
|
+
await fire(host, "tool_execution_end", {
|
|
503
|
+
type: "tool_execution_end",
|
|
504
|
+
toolCallId: "t1",
|
|
505
|
+
toolName: "read",
|
|
506
|
+
result: {},
|
|
507
|
+
isError: false,
|
|
508
|
+
}, ctx);
|
|
509
|
+
|
|
510
|
+
// Budget flip arms slimming; the boundary attaches the drafts.
|
|
511
|
+
// Budget-limited goals stop auto-continuing (omp semantics), so the
|
|
512
|
+
// boundary persists the edits without a next request.
|
|
513
|
+
const result = fireBeforeSettle(host, ctx);
|
|
514
|
+
expect(result?.continue).toBeUndefined();
|
|
515
|
+
const edits = (result?.entries ?? []).filter((e) => e.type === "context_edit");
|
|
516
|
+
expect(edits).toEqual([{ type: "context_edit", targetId: "t-old", replacement: null }]);
|
|
517
|
+
|
|
518
|
+
// Once per goal: the next settle carries no further slimming drafts.
|
|
519
|
+
entries.push({
|
|
520
|
+
type: "message",
|
|
521
|
+
id: "m2",
|
|
522
|
+
message: { role: "assistant", stopReason: "toolUse", usage: { input: 50, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
523
|
+
});
|
|
524
|
+
await fire(host, "tool_execution_end", {
|
|
525
|
+
type: "tool_execution_end",
|
|
526
|
+
toolCallId: "t2",
|
|
527
|
+
toolName: "read",
|
|
528
|
+
result: {},
|
|
529
|
+
isError: false,
|
|
530
|
+
}, ctx);
|
|
531
|
+
expect(fireBeforeSettle(host, ctx)).toBeUndefined();
|
|
532
|
+
vi.unstubAllEnvs();
|
|
533
|
+
});
|
|
534
|
+
|
|
535
|
+
it("context slimming stays off by default (opt-in via PI_GOAL_SLIM_ON_BUDGET)", async () => {
|
|
536
|
+
const host = fakeHost();
|
|
537
|
+
defaultExport(host.pi);
|
|
538
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
539
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
540
|
+
await host.commands.goal!.handler("budget 10", createContext(host));
|
|
541
|
+
|
|
542
|
+
const entries: unknown[] = [
|
|
543
|
+
{ type: "message", id: "t-old", message: { role: "toolResult" } },
|
|
544
|
+
{ type: "message", id: "u1", message: { role: "user" } },
|
|
545
|
+
{ type: "message", id: "t-new", message: { role: "toolResult" } },
|
|
546
|
+
];
|
|
547
|
+
const ctx = createContext(host, { entries });
|
|
548
|
+
|
|
549
|
+
// Explicitly opt out (order-independent against leaking stubs).
|
|
550
|
+
vi.stubEnv("PI_GOAL_SLIM_ON_BUDGET", "");
|
|
551
|
+
await fire(host, "turn_start", { type: "turn_start", turnIndex: 0, timestamp: 0 }, ctx);
|
|
552
|
+
entries.push({
|
|
553
|
+
type: "message",
|
|
554
|
+
id: "m1",
|
|
555
|
+
message: { role: "assistant", stopReason: "toolUse", usage: { input: 25, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
556
|
+
});
|
|
557
|
+
await fire(host, "tool_execution_end", {
|
|
558
|
+
type: "tool_execution_end",
|
|
559
|
+
toolCallId: "t1",
|
|
560
|
+
toolName: "read",
|
|
561
|
+
result: {},
|
|
562
|
+
isError: false,
|
|
563
|
+
}, ctx);
|
|
564
|
+
|
|
565
|
+
// Budget flip steers but arms no slimming drafts; budget-limited goals
|
|
566
|
+
// do not auto-continue, so the boundary settles empty.
|
|
567
|
+
expect(fireBeforeSettle(host, ctx)).toBeUndefined();
|
|
568
|
+
vi.unstubAllEnvs();
|
|
569
|
+
});
|
|
570
|
+
|
|
431
571
|
it("/guided-goal queues a hidden interview kickoff", async () => {
|
|
432
572
|
const host = fakeHost();
|
|
433
573
|
defaultExport(host.pi);
|
|
@@ -488,7 +628,8 @@ describe("pi-goal extension wiring", () => {
|
|
|
488
628
|
|
|
489
629
|
await fire(host, "agent_start", { type: "agent_start" });
|
|
490
630
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
|
|
491
|
-
|
|
631
|
+
// Modal open at settle: the boundary withholds the continuation.
|
|
632
|
+
expect(fireBeforeSettle(host)?.continue).toBeUndefined();
|
|
492
633
|
|
|
493
634
|
// Dialog closes while idle: the withheld continuation is scheduled now.
|
|
494
635
|
const idleCtx = createContext(host, { idle: true });
|
|
@@ -517,7 +658,8 @@ describe("pi-goal extension wiring", () => {
|
|
|
517
658
|
expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
|
|
518
659
|
|
|
519
660
|
await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
|
|
520
|
-
|
|
661
|
+
// Modal already closed: the boundary itself continues the run.
|
|
662
|
+
expect(fireBeforeSettle(host)?.continue).toBe(true);
|
|
521
663
|
});
|
|
522
664
|
|
|
523
665
|
it("refreshes the status on agent_settled (status integrations per docs)", async () => {
|
|
@@ -632,4 +774,77 @@ describe("pi-goal extension wiring", () => {
|
|
|
632
774
|
expect(component.text).toContain("6K (no budget) tokens");
|
|
633
775
|
expect(component.text).toContain("20s");
|
|
634
776
|
});
|
|
777
|
+
|
|
778
|
+
it("pauses the goal instead of continuing when the run ends with a provider error", async () => {
|
|
779
|
+
const host = fakeHost();
|
|
780
|
+
defaultExport(host.pi);
|
|
781
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
782
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
783
|
+
host.sentMessages.length = 0;
|
|
784
|
+
|
|
785
|
+
await fire(host, "agent_start", { type: "agent_start" });
|
|
786
|
+
await fire(host, "agent_end", {
|
|
787
|
+
type: "agent_end",
|
|
788
|
+
messages: [
|
|
789
|
+
{ role: "assistant", stopReason: "error", errorMessage: "429 rate limit exceeded", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
790
|
+
],
|
|
791
|
+
});
|
|
792
|
+
|
|
793
|
+
// Paused and persisted, with a classified notice; no continuation queued.
|
|
794
|
+
const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
|
|
795
|
+
expect(pauseEntry?.data).toMatchObject({ enabled: false, goal: { status: "paused" } });
|
|
796
|
+
expect(host.notifications.some((n) => n.includes("rate limits"))).toBe(true);
|
|
797
|
+
expect(fireBeforeSettle(host)?.continue).toBeUndefined();
|
|
798
|
+
});
|
|
799
|
+
|
|
800
|
+
it("pauses with a generic notice on non-usage errors", async () => {
|
|
801
|
+
const host = fakeHost();
|
|
802
|
+
defaultExport(host.pi);
|
|
803
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
804
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
805
|
+
|
|
806
|
+
await fire(host, "agent_start", { type: "agent_start" });
|
|
807
|
+
await fire(host, "agent_end", {
|
|
808
|
+
type: "agent_end",
|
|
809
|
+
messages: [
|
|
810
|
+
{ role: "assistant", stopReason: "error", errorMessage: "socket hang up", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
|
|
811
|
+
],
|
|
812
|
+
});
|
|
813
|
+
|
|
814
|
+
expect(host.notifications.some((n) => n.includes("ended with an error"))).toBe(true);
|
|
815
|
+
});
|
|
816
|
+
|
|
817
|
+
it("context pruning keeps only the newest goal messages and drops stale continuations", async () => {
|
|
818
|
+
const host = fakeHost();
|
|
819
|
+
defaultExport(host.pi);
|
|
820
|
+
await fire(host, "session_start", { type: "session_start", reason: "startup" });
|
|
821
|
+
await host.commands.goal!.handler("Do the thing", createContext(host));
|
|
822
|
+
const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
|
|
823
|
+
|
|
824
|
+
const custom = (customType: string, details?: unknown) => ({ role: "custom", customType, display: false, details });
|
|
825
|
+
const messages = [
|
|
826
|
+
{ role: "user", content: "hi" },
|
|
827
|
+
custom("goal-mode-context"), // stale context
|
|
828
|
+
custom("goal-continuation", { goalId: "old-goal" }), // stale goal id
|
|
829
|
+
custom("goal-budget-limit"),
|
|
830
|
+
custom("goal-mode-context"), // newest context: kept
|
|
831
|
+
custom("goal-continuation", { goalId }), // newest for the active goal: kept
|
|
832
|
+
{ role: "assistant", content: "working" },
|
|
833
|
+
custom("goal-continuation", { goalId: "old-goal" }), // another stale: dropped
|
|
834
|
+
];
|
|
835
|
+
|
|
836
|
+
const handler = host.handlers.get("context")?.[0]!;
|
|
837
|
+
const result = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
|
|
838
|
+
const kept = result.messages.filter((m) => (m as { role?: string }).role === "custom");
|
|
839
|
+
expect(kept.map((m) => (m as { customType: string }).customType)).toEqual([
|
|
840
|
+
"goal-budget-limit",
|
|
841
|
+
"goal-mode-context",
|
|
842
|
+
"goal-continuation",
|
|
843
|
+
]);
|
|
844
|
+
|
|
845
|
+
// Once no goal is active, every continuation is dropped.
|
|
846
|
+
await host.commands.goal!.handler("drop", createContext(host));
|
|
847
|
+
const result2 = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
|
|
848
|
+
expect(result2.messages.filter((m) => (m as { customType?: string }).customType === "goal-continuation")).toHaveLength(0);
|
|
849
|
+
});
|
|
635
850
|
});
|
package/test/runtime.test.ts
CHANGED
|
@@ -468,4 +468,22 @@ describe("goal runtime", () => {
|
|
|
468
468
|
const second = await harness.runtime.createGoal({ objective: "Two" });
|
|
469
469
|
expect(first.goal.id).not.toBe(second.goal.id);
|
|
470
470
|
});
|
|
471
|
+
|
|
472
|
+
it("rejects objectives beyond the character cap on create and replace", async () => {
|
|
473
|
+
const harness = createHarness();
|
|
474
|
+
const oversized = "x".repeat(4_001);
|
|
475
|
+
await expect(harness.runtime.createGoal({ objective: oversized })).rejects.toThrow(/too long.*4,000/s);
|
|
476
|
+
// Boundary: exactly at the cap is fine.
|
|
477
|
+
await harness.runtime.createGoal({ objective: "x".repeat(4_000) });
|
|
478
|
+
// Replace path enforces the same cap.
|
|
479
|
+
await expect(harness.runtime.replaceGoal({ objective: oversized })).rejects.toThrow(/too long/);
|
|
480
|
+
});
|
|
481
|
+
|
|
482
|
+
it("counts characters as code points, not UTF-16 units", async () => {
|
|
483
|
+
const harness = createHarness();
|
|
484
|
+
// Astral emoji: 1 code point, 2 UTF-16 units. 2_001 emoji is 4_002 UTF-16
|
|
485
|
+
// units (a naive .length cap would reject it) but 2_001 code points: kept.
|
|
486
|
+
await expect(harness.runtime.createGoal({ objective: "😀".repeat(2_001) })).resolves.toBeDefined();
|
|
487
|
+
await expect(harness.runtime.createGoal({ objective: "😀".repeat(4_001) })).rejects.toThrow(/too long/);
|
|
488
|
+
});
|
|
471
489
|
});
|
package/test/tool.test.ts
CHANGED
|
@@ -90,8 +90,14 @@ function stubEvaluator(outcome: GoalEvaluatorOutcome) {
|
|
|
90
90
|
function createTestTool(
|
|
91
91
|
harness: ReturnType<typeof createRuntimeHarness>,
|
|
92
92
|
evaluator: (request: GoalEvaluatorRequest, opts: { cwd?: string }) => Promise<GoalEvaluatorOutcome>,
|
|
93
|
+
onActivated?: () => Promise<void>,
|
|
93
94
|
) {
|
|
94
|
-
return createGoalTool({
|
|
95
|
+
return createGoalTool({
|
|
96
|
+
getRuntime: () => harness.runtime,
|
|
97
|
+
getState: harness.getState,
|
|
98
|
+
runEvaluator: evaluator,
|
|
99
|
+
onActivated,
|
|
100
|
+
});
|
|
95
101
|
}
|
|
96
102
|
|
|
97
103
|
// ---------------------------------------------------------------------------
|
|
@@ -99,6 +105,25 @@ function createTestTool(
|
|
|
99
105
|
// ---------------------------------------------------------------------------
|
|
100
106
|
|
|
101
107
|
describe("goal tool", () => {
|
|
108
|
+
it("onActivated hook fires on create and resume (mid-run context injection)", async () => {
|
|
109
|
+
const harness = createRuntimeHarness();
|
|
110
|
+
const activations: string[] = [];
|
|
111
|
+
const tool = createTestTool(
|
|
112
|
+
harness,
|
|
113
|
+
async () => ({ status: "confirmed", reason: "ok" }),
|
|
114
|
+
async () => {
|
|
115
|
+
activations.push("activated");
|
|
116
|
+
},
|
|
117
|
+
);
|
|
118
|
+
await executeTool(tool, { op: "create", objective: "Ship it" });
|
|
119
|
+
expect(activations).toHaveLength(1);
|
|
120
|
+
await executeTool(tool, { op: "resume" });
|
|
121
|
+
expect(activations).toHaveLength(2);
|
|
122
|
+
// get does not re-activate.
|
|
123
|
+
await executeTool(tool, { op: "get" });
|
|
124
|
+
expect(activations).toHaveLength(2);
|
|
125
|
+
});
|
|
126
|
+
|
|
102
127
|
it("create starts a goal and returns the objective/status text", async () => {
|
|
103
128
|
const harness = createRuntimeHarness();
|
|
104
129
|
const tool = createTestTool(harness, async () => ({ status: "confirmed", reason: "verified" }));
|