@fyeeme/pi-goal 1.0.2 → 1.0.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,24 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [1.0.4] - 2026-10-01
6
+
7
+ ### Added
8
+
9
+ - Objective length cap (`MAX_OBJECTIVE_CHARS`, 4,000 code points) enforced by the runtime on `create` and `replace` — the objective is re-injected into context on every continuation and passed to the evaluator subprocess, so an oversized one silently burns budget every turn; the error guides long instructions into a referenced file. Counting is code-point-based, not UTF-16 units. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
10
+
11
+ ### Fixed
12
+
13
+ - Print/json/rpc modes could never create a goal: `session_start` removed the `goal` tool from the active set when no goal existed (omp sdk.ts parity), but non-interactive modes have no `/goal` or `/guided-goal` command to re-arm it — leaving the model unable to start a goal at all (found via live goal-mode testing). The tool is now only removed in TUI mode.
14
+ - The run that creates/resumes a goal via the tool never saw the goal context prompt: `before_agent_start` injection only fires on the next agent run, and the command-path steer was not wired to the tool path (found via live goal-mode testing). The tool now fires an `onActivated` hook so the host injects the goal context into the current run as a steer.
15
+ - The evaluator subprocess ran with full extension discovery: any other installed goal extension (with an active-goal system prompt) would pollute the judge that is supposed to be independent, and the startup overhead contributed to live 300s timeouts. It now runs lean (`--no-extensions --no-skills`), the default timeout is raised to 600s, and `GOAL_EVALUATOR_TIMEOUT_MS` overrides it.
16
+
17
+ ### Changed
18
+
19
+ - Context hygiene: hidden goal messages no longer accumulate in the LLM's view. A `context` handler keeps only the newest `goal-mode-context` and `goal-budget-limit` messages plus the newest `goal-continuation` stamped for the currently active goal id (`details.goalId`); stale ones — including all continuations once no goal is active — are dropped from the model's view, not from the transcript. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
20
+ - Run-error handling: when a run ends with an assistant `stopReason: "error"`, the active goal now pauses (persisted) instead of letting the continuation loop fire into a likely retry loop, with a classified notice — provider usage/rate/quota/limit errors read differently from generic faults. Abort behavior is unchanged. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
21
+ - Peer dependency floor raised to `@earendil-works/pi-coding-agent >= 0.99.0`; dev toolchain pinned to 0.99.2 (typecheck and tests pass against 0.99.2 unchanged).
22
+
5
23
  ## [1.0.2] - 2026-09-16
6
24
 
7
25
  ### Changed
package/index.ts CHANGED
@@ -27,12 +27,18 @@
27
27
  * goal_updated session event → pi.events.emit("goal_updated")
28
28
  * sendHiddenMessage (budget steer) → pi.sendMessage display:false
29
29
  * prompt-time prependMessages goal context
30
- * → before_agent_start message
31
- * injection (hidden custom message)
30
+ * → context_with_system per-request
31
+ * injection (0.87.0): survives
32
+ * compaction, supersedes transcript
33
+ * copies instead of accumulating
32
34
  * #scheduleGoalContinuation 800ms TUI timer
33
- * → followUp delivery + triggerTurn
34
- * (editor-empty guard dropped: pi
35
- * extensions cannot read the editor)
35
+ * → agent_before_settle actionable
36
+ * boundary (0.87.0):
37
+ * { entries: [continuation draft],
38
+ * continue: true } — exactly one
39
+ * next provider request; sendMessage
40
+ * fallback kept for continuations
41
+ * withheld after settlement
36
42
  * setActiveToolsByName → pi.setActiveTools
37
43
  * status-line segment → ctx.ui.setStatus("goal", ...)
38
44
  * settings goal.enabled / continuationModes / statusInFooter
@@ -59,7 +65,12 @@
59
65
  import { readFileSync } from "node:fs";
60
66
  import * as path from "node:path";
61
67
  import { fileURLToPath } from "node:url";
62
- import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent";
68
+ import type {
69
+ ContextEditEntryDraft,
70
+ ExtensionAPI,
71
+ ExtensionContext,
72
+ SessionBoundaryDraft,
73
+ } from "@earendil-works/pi-coding-agent";
63
74
  import { Text } from "@earendil-works/pi-tui";
64
75
  import {
65
76
  createGoalCommand,
@@ -97,6 +108,7 @@ const goalModeContextPrompt = readFileSync(
97
108
  interface EntryMessageLike {
98
109
  role?: string;
99
110
  stopReason?: string;
111
+ errorMessage?: string;
100
112
  usage?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined;
101
113
  }
102
114
 
@@ -104,6 +116,34 @@ interface EntryUsageLike {
104
116
  usage?: EntryMessageLike["usage"];
105
117
  }
106
118
 
119
+ /** Budget slimming drafts (opt-in via PI_GOAL_SLIM_ON_BUDGET=1, attached once
120
+ * per goal at the budget-limited flip): omit toolResult entries that precede
121
+ * the newest real user message from future provider context via append-only
122
+ * context-edit drafts — superseded turns' work products. Raw history, usage
123
+ * and UI history stay untouched (context_edit contract). Pure: returns the
124
+ * drafts; the settle boundary attaches them. */
125
+ export function contextSlimDrafts(branch: unknown[]): ContextEditEntryDraft[] {
126
+ const drafts: ContextEditEntryDraft[] = [];
127
+ let keepFrom = -1;
128
+ for (let i = branch.length - 1; i >= 0; i--) {
129
+ const entry = branch[i] as { type?: string; message?: { role?: string } } | undefined;
130
+ if (entry?.type === "message" && entry.message?.role === "user") {
131
+ keepFrom = i;
132
+ break;
133
+ }
134
+ }
135
+ if (keepFrom <= 0) return drafts;
136
+ const seen = new Set<string>();
137
+ for (let i = 0; i < keepFrom; i++) {
138
+ const entry = branch[i] as { type?: string; id?: string; message?: { role?: string } } | undefined;
139
+ if (entry?.type !== "message" || entry.message?.role !== "toolResult") continue;
140
+ if (!entry.id || seen.has(entry.id)) continue;
141
+ seen.add(entry.id);
142
+ drafts.push({ type: "context_edit", targetId: entry.id, replacement: null });
143
+ }
144
+ return drafts;
145
+ }
146
+
107
147
  export default function piGoalExtension(pi: ExtensionAPI): void {
108
148
  // ------------------------------------------------------------------
109
149
  // Closure state (omp session fields)
@@ -125,6 +165,10 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
125
165
  /** A blocking ctx.ui dialog is open (ui_prompt_start/end). While open, no
126
166
  * continuation may fire: the modal would hide a turn starting behind it. */
127
167
  let uiPromptOpen = false;
168
+ /** Goal id whose budget-limited flip already armed context slimming. */
169
+ let slimmedBudgetGoalId: string | undefined;
170
+ /** Context-slimming drafts pending attachment at the next settle boundary. */
171
+ let pendingSlim = false;
128
172
 
129
173
  // ------------------------------------------------------------------
130
174
  // Usage accounting (mirror of pi getSessionStats().tokens sums)
@@ -256,6 +300,14 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
256
300
  return;
257
301
  }
258
302
  goalState = state;
303
+ if (
304
+ state?.goal.status === "budget-limited" &&
305
+ process.env.PI_GOAL_SLIM_ON_BUDGET === "1" &&
306
+ slimmedBudgetGoalId !== state.goal.id
307
+ ) {
308
+ slimmedBudgetGoalId = state.goal.id;
309
+ pendingSlim = true;
310
+ }
259
311
  if (!state?.enabled) {
260
312
  continuationInFlight = false;
261
313
  }
@@ -340,6 +392,23 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
340
392
  await exitGoalMode({ reason: "paused" });
341
393
  }
342
394
 
395
+ /** A provider/agent error ended the run (borrowed from mitsuhiko/agent-stuff
396
+ * goal.ts): pause the active goal instead of letting scheduleContinuation
397
+ * fire into a likely retry loop, and classify the failure for the user.
398
+ * Usage/rate/quota errors read differently from generic faults. */
399
+ async function pauseOnRunError(messages: unknown[]): Promise<void> {
400
+ if (!goalState?.enabled || goalState.goal.status !== "active") return;
401
+ const errorMessage = lastAssistantErrorMessage(messages) ?? "";
402
+ const usageLimited = /\b(usage|rate|quota|limit)\b/i.test(errorMessage);
403
+ await runtime.pauseGoal();
404
+ notify(
405
+ usageLimited
406
+ ? "Goal paused: the last turn hit provider usage/rate limits. /goal resume when ready."
407
+ : "Goal paused: the last turn ended with an error. /goal resume to continue.",
408
+ "error",
409
+ );
410
+ }
411
+
343
412
  async function dropGoal(): Promise<void> {
344
413
  if (!goalState) {
345
414
  notify("No goal to drop.", "warning");
@@ -398,7 +467,9 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
398
467
  return renderTemplate(goalModeContextPrompt, { goalContext: content, todoContext });
399
468
  }
400
469
 
401
- /** omp #scheduleGoalContinuation (TUI timer replaced by followUp delivery). */
470
+ /** omp #scheduleGoalContinuation. Post-settlement fallback path only: the
471
+ * primary continuation decision lives in the agent_before_settle boundary
472
+ * below (0.87.0). */
402
473
  function scheduleContinuation(): void {
403
474
  if (!goalState?.enabled || goalState.goal.status !== "active") return;
404
475
  if (suppressNextContinuation) return;
@@ -408,11 +479,33 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
408
479
  if (!prompt) return;
409
480
  continuationInFlight = true;
410
481
  pi.sendMessage(
411
- { customType: "goal-continuation", content: prompt, display: false },
482
+ // details.goalId keys context pruning (see the "context" handler below):
483
+ // only the newest continuation of the CURRENTLY active goal survives.
484
+ { customType: "goal-continuation", content: prompt, display: false, details: { goalId: goalState.goal.id } },
412
485
  { triggerTurn: true, deliverAs: "followUp" },
413
486
  );
414
487
  }
415
488
 
489
+ /** omp #scheduleGoalContinuation decision chain, settled into a draft:
490
+ * guards pass → one custom_message draft carrying the continuation
491
+ * prompt (details.goalId keys context pruning). */
492
+ function continuationDraft(ctx: ExtensionContext): SessionBoundaryDraft | undefined {
493
+ if (!goalState?.enabled || goalState.goal.status !== "active") return undefined;
494
+ if (suppressNextContinuation) return undefined;
495
+ if (uiPromptOpen) return undefined; // modal open: never start a turn behind it
496
+ if (ctx.hasPendingMessages()) return undefined;
497
+ const prompt = runtime.buildContinuationPrompt();
498
+ if (!prompt) return undefined;
499
+ continuationInFlight = true;
500
+ return {
501
+ type: "custom_message",
502
+ customType: "goal-continuation",
503
+ content: prompt,
504
+ display: false,
505
+ details: { goalId: goalState.goal.id },
506
+ };
507
+ }
508
+
416
509
  // ------------------------------------------------------------------
417
510
  // Tool + command registration
418
511
  // ------------------------------------------------------------------
@@ -425,6 +518,11 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
425
518
  // fall back to the process cwd.
426
519
  runEvaluator: (request, opts) =>
427
520
  runGoalEvaluator(request, { cwd: opts.cwd ?? currentCtx?.cwd ?? process.cwd(), signal: opts.signal }),
521
+ // Mid-run activation (tool create/resume): inject the goal context into
522
+ // the CURRENT run as a steer — before_agent_start only covers the next run.
523
+ onActivated: async () => {
524
+ if (currentCtx && !currentCtx.isIdle()) await sendGoalModeContext("steer");
525
+ },
428
526
  };
429
527
  pi.registerTool(createGoalTool(goalToolDeps));
430
528
 
@@ -501,11 +599,16 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
501
599
  const restored = restoreGoalFromEntries(branch);
502
600
  if (!restored) {
503
601
  goalState = undefined;
504
- // omp sdk.ts excludes the goal tool from the initial set; mirror that
505
- // when this session has no goal to manage.
506
- const active = pi.getActiveTools();
507
- if (active.includes("goal")) {
508
- pi.setActiveTools(active.filter((name) => name !== "goal"));
602
+ // omp sdk.ts excludes the goal tool from the initial set; mirror that in
603
+ // TUI where /goal and /guided-goal re-arm it. In non-interactive modes
604
+ // (print/json/rpc) no slash command exists, so removing the tool would
605
+ // leave the model unable to create a goal at all (found via live
606
+ // goal-mode testing): keep it active there.
607
+ if (ctx.mode === "tui") {
608
+ const active = pi.getActiveTools();
609
+ if (active.includes("goal")) {
610
+ pi.setActiveTools(active.filter((name) => name !== "goal"));
611
+ }
509
612
  }
510
613
  updateStatus();
511
614
  return;
@@ -554,12 +657,59 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
554
657
  }
555
658
  });
556
659
 
557
- pi.on("before_agent_start", (_event, ctx) => {
660
+ // Goal context injection (0.87.0 context_with_system): runs after `context`
661
+ // handlers on the full transcript including system messages; the result is
662
+ // sent verbatim. Per-request injection survives compaction (re-applied every
663
+ // call) and the fresh copy supersedes any steer-injected transcript copies.
664
+ pi.on("context_with_system", (event, ctx) => {
558
665
  currentCtx = ctx;
559
666
  const content = buildGoalModeMessage();
560
667
  if (!content) return undefined;
668
+ const messages = event.messages.filter((message) => {
669
+ const msg = message as { role?: string; customType?: string } | undefined;
670
+ return !(msg?.role === "custom" && msg.customType === "goal-mode-context");
671
+ });
672
+ messages.push({
673
+ role: "custom",
674
+ customType: "goal-mode-context",
675
+ content,
676
+ display: false,
677
+ } as (typeof event.messages)[number]);
678
+ return { messages };
679
+ });
680
+
681
+ // Context hygiene (borrowed from mitsuhiko/agent-stuff goal.ts): hidden goal
682
+ // messages otherwise accumulate one per run/continuation and burn context
683
+ // every LLM call. Keep only the newest goal-mode-context and
684
+ // goal-budget-limit, plus the newest continuation stamped for the currently
685
+ // active goal id; stale ones (including all continuations once no goal is
686
+ // active) are dropped from the model's view, not from the transcript.
687
+ pi.on("context", (event) => {
688
+ const activeGoalId = goalState?.enabled && goalState.goal.status === "active" ? goalState.goal.id : undefined;
689
+ let lastContext = -1;
690
+ let lastBudget = -1;
691
+ let lastContinuation = -1;
692
+ for (let i = 0; i < event.messages.length; i++) {
693
+ const msg = event.messages[i] as
694
+ | { role?: string; customType?: string; details?: { goalId?: string } }
695
+ | undefined;
696
+ if (msg?.role !== "custom") continue;
697
+ if (msg.customType === "goal-mode-context") lastContext = i;
698
+ else if (msg.customType === "goal-budget-limit") lastBudget = i;
699
+ else if (msg.customType === "goal-continuation" && activeGoalId !== undefined && msg.details?.goalId === activeGoalId) {
700
+ lastContinuation = i;
701
+ }
702
+ }
703
+ if (lastContext === -1 && lastBudget === -1 && lastContinuation === -1) return undefined;
561
704
  return {
562
- message: { customType: "goal-mode-context", content, display: false },
705
+ messages: event.messages.filter((message, index) => {
706
+ const msg = message as { role?: string; customType?: string } | undefined;
707
+ if (msg?.role !== "custom") return true;
708
+ if (msg.customType === "goal-mode-context") return index === lastContext;
709
+ if (msg.customType === "goal-budget-limit") return index === lastBudget;
710
+ if (msg.customType === "goal-continuation") return index === lastContinuation;
711
+ return true;
712
+ }),
563
713
  };
564
714
  });
565
715
 
@@ -567,11 +717,12 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
567
717
  currentCtx = ctx;
568
718
  // omp separates onAgentEnd (session) from continuation scheduling
569
719
  // (interactive-mode); both subscribe to the same end-of-run moment.
570
- const aborted = lastAssistantStopReason(event.messages) === "aborted";
571
- if (aborted) {
720
+ const stopReason = lastAssistantStopReason(event.messages);
721
+ if (stopReason === "aborted") {
572
722
  await runtime.onTaskAborted({ reason: "interrupted" });
573
723
  } else {
574
724
  await runtime.onAgentEnd({ currentUsage: currentUsage(ctx) });
725
+ if (stopReason === "error") await pauseOnRunError(event.messages);
575
726
  }
576
727
 
577
728
  if (continuationInFlight) {
@@ -583,7 +734,31 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
583
734
  return;
584
735
  }
585
736
  updateStatus();
586
- scheduleContinuation();
737
+ });
738
+
739
+ // omp #scheduleGoalContinuation migrated to the 0.87.0 actionable boundary:
740
+ // at settle time decide ONCE whether exactly one more provider request
741
+ // should run, attaching the continuation prompt as a custom_message draft.
742
+ // No timer, no followUp scheduling — pi owns the request. Guards mirror the
743
+ // old scheduleContinuation chain; continuations withheld after settlement
744
+ // (modal closed late, budget set while idle) still recover via the
745
+ // sendMessage fallback in scheduleContinuation().
746
+ pi.on("agent_before_settle", (_event, ctx) => {
747
+ currentCtx = ctx;
748
+ // Context slimming (opt-in): attach pending context_edit drafts first —
749
+ // they persist even when the continuation itself is withheld.
750
+ const drafts: SessionBoundaryDraft[] = [];
751
+ if (pendingSlim) {
752
+ pendingSlim = false;
753
+ drafts.push(...contextSlimDrafts(ctx.sessionManager.getBranch()));
754
+ }
755
+ const continuation = continuationDraft(ctx);
756
+ if (continuation) drafts.push(continuation);
757
+ if (drafts.length === 0) return undefined;
758
+ // Absent `continue` = no opinion on auto-continuation (budget-limited
759
+ // goals settle without one); `continue: true` guarantees exactly one
760
+ // next provider request.
761
+ return continuation !== undefined ? { entries: drafts, continue: true } : { entries: drafts };
587
762
  });
588
763
 
589
764
  pi.on("turn_end", (_event, ctx) => {
@@ -665,6 +840,14 @@ function lastAssistantStopReason(messages: unknown[]): string | undefined {
665
840
  return undefined;
666
841
  }
667
842
 
843
+ function lastAssistantErrorMessage(messages: unknown[]): string | undefined {
844
+ for (let i = messages.length - 1; i >= 0; i--) {
845
+ const message = messages[i] as EntryMessageLike | undefined;
846
+ if (message?.role === "assistant") return message.errorMessage;
847
+ }
848
+ return undefined;
849
+ }
850
+
668
851
  export type { GoalRuntimeHost } from "./src/runtime.ts";
669
852
  // Re-exported for consumers that compose the pieces directly (tests, tools).
670
853
  export { GoalRuntime } from "./src/runtime.ts";
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@fyeeme/pi-goal",
3
- "version": "1.0.2",
4
- "description": "Goal mode for pi \u2014 one persistent autonomous objective looped until verified success: goal tool (create/get/complete/resume/drop), token/time budget accounting with budget-limit steering, automatic continuation turns, /goal and /guided-goal commands, and a goal_updated event bus for other extensions.",
3
+ "version": "1.0.4",
4
+ "description": "Goal mode for pi — one persistent autonomous objective looped until verified success: goal tool (create/get/complete/resume/drop), token/time budget accounting with budget-limit steering, automatic continuation turns, /goal and /guided-goal commands, and a goal_updated event bus for other extensions.",
5
5
  "type": "module",
6
6
  "license": "MIT",
7
7
  "author": "fyeeme",
@@ -47,17 +47,17 @@
47
47
  "typecheck": "tsc"
48
48
  },
49
49
  "peerDependencies": {
50
- "@earendil-works/pi-ai": ">=0.84.4",
51
- "@earendil-works/pi-coding-agent": ">=0.84.4",
52
- "@earendil-works/pi-tui": ">=0.84.4",
50
+ "@earendil-works/pi-ai": ">=0.99.0",
51
+ "@earendil-works/pi-coding-agent": ">=0.99.0",
52
+ "@earendil-works/pi-tui": ">=0.99.0",
53
53
  "typebox": ">=1.0.0",
54
- "@earendil-works/pi-agent-core": "0.84.4"
54
+ "@earendil-works/pi-agent-core": "0.99.2"
55
55
  },
56
56
  "devDependencies": {
57
- "@earendil-works/pi-agent-core": "0.84.4",
58
- "@earendil-works/pi-ai": "0.84.4",
59
- "@earendil-works/pi-coding-agent": "0.84.4",
60
- "@earendil-works/pi-tui": "0.84.4",
57
+ "@earendil-works/pi-agent-core": "0.99.2",
58
+ "@earendil-works/pi-ai": "0.99.2",
59
+ "@earendil-works/pi-coding-agent": "0.99.2",
60
+ "@earendil-works/pi-tui": "0.99.2",
61
61
  "@types/node": "22.19.19",
62
62
  "typescript": "5.9.3",
63
63
  "vitest": "3.2.7"
package/src/evaluator.ts CHANGED
@@ -34,8 +34,16 @@ const evaluatorCompletePrompt = readFileSync(path.join(promptsDir, "evaluator-co
34
34
  const evaluatorImpossiblePrompt = readFileSync(path.join(promptsDir, "evaluator-impossible.md"), "utf8");
35
35
 
36
36
  /** Hard cap on one evaluator run — the grounded check may run test suites,
37
- * so this is deliberately generous; anything longer has failed. */
38
- export const EVALUATOR_TIMEOUT_MS = 300_000;
37
+ * so this is deliberately generous; anything longer has failed. Each LLM
38
+ * call on slow providers can take 20s+, and the evaluator is a multi-step
39
+ * agent (inspect repo, run checks, emit JSON), so the default needs real
40
+ * headroom. Override with GOAL_EVALUATOR_TIMEOUT_MS. */
41
+ export const EVALUATOR_TIMEOUT_MS = 600_000;
42
+
43
+ function evaluatorTimeout(): number {
44
+ const raw = Number.parseInt(process.env.GOAL_EVALUATOR_TIMEOUT_MS ?? "", 10);
45
+ return Number.isInteger(raw) && raw > 0 ? raw : EVALUATOR_TIMEOUT_MS;
46
+ }
39
47
 
40
48
  /** Argv-safety caps (kernel MAX_ARG_STRLEN ≈ 128KB; these keep the prompt
41
49
  * argument far under it even for verbose objectives/audits). */
@@ -118,6 +126,16 @@ export function buildEvaluatorPrompt(request: GoalEvaluatorRequest): string {
118
126
  });
119
127
  }
120
128
 
129
+ /** Build the evaluator argv: a one-shot headless run. Lean flags matter here:
130
+ * without them the nested pi loads the user's full extension set — including
131
+ * any OTHER goal extension, whose active-goal system prompt would pollute the
132
+ * very evaluator that is supposed to judge the goal independently — and its
133
+ * startup cost pushes long verification runs over the timeout (observed
134
+ * live: 300s timeout hit while the evaluator was still working). */
135
+ function evaluatorInvocation(prompt: string): { command: string; args: string[] } {
136
+ return getPiInvocation(["-p", "--no-session", "--no-extensions", "--no-skills", prompt]);
137
+ }
138
+
121
139
  /** Extract the first balanced JSON object from evaluator output. Handles the
122
140
  * clean case, code-fenced output, and prose-wrapped JSON; anything else is
123
141
  * unparseable. Never throws. */
@@ -181,10 +199,10 @@ export async function runGoalEvaluator(
181
199
  ): Promise<GoalEvaluatorOutcome> {
182
200
  const prompt = buildEvaluatorPrompt(request);
183
201
  const spawn = opts.spawn ?? defaultSpawn;
184
- const timeoutMs = opts.timeoutMs ?? EVALUATOR_TIMEOUT_MS;
202
+ const timeoutMs = opts.timeoutMs ?? evaluatorTimeout();
185
203
  let stdout: string;
186
204
  try {
187
- stdout = await spawn(getPiInvocation(["-p", "--no-session", prompt]), {
205
+ stdout = await spawn(evaluatorInvocation(prompt), {
188
206
  cwd: opts.cwd,
189
207
  timeoutMs,
190
208
  signal: opts.signal,
package/src/runtime.ts CHANGED
@@ -128,6 +128,25 @@ export function completionBudgetReport(goal: Goal): string | null {
128
128
  return `Goal achieved. Report final budget usage to the user: ${parts.join("; ")}.`;
129
129
  }
130
130
 
131
+ /** Objective length cap. The objective is re-injected into context on every
132
+ * continuation and passed to the evaluator subprocess, so an oversized one
133
+ * silently burns budget every turn. Borrowed from mitsuhiko/agent-stuff
134
+ * goal.ts (MAX_OBJECTIVE_CHARS). */
135
+ export const MAX_OBJECTIVE_CHARS = 4_000;
136
+
137
+ function charCount(value: string): number {
138
+ return [...value].length;
139
+ }
140
+
141
+ function validateObjective(objective: string): void {
142
+ const count = charCount(objective);
143
+ if (count > MAX_OBJECTIVE_CHARS) {
144
+ throw new Error(
145
+ `Goal objective is too long: ${count.toLocaleString()} characters. Limit: ${MAX_OBJECTIVE_CHARS.toLocaleString()} characters. Put longer instructions in a file and reference that file in the objective.`,
146
+ );
147
+ }
148
+ }
149
+
131
150
  function validateTokenBudget(tokenBudget: number | undefined): void {
132
151
  if (tokenBudget !== undefined && (!Number.isInteger(tokenBudget) || tokenBudget <= 0)) {
133
152
  throw new Error("goal token_budget must be a positive integer when provided");
@@ -408,6 +427,7 @@ export class GoalRuntime {
408
427
  async createGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
409
428
  const objective = input.objective.trim();
410
429
  if (!objective) throw new Error("objective is required when op=create");
430
+ validateObjective(objective);
411
431
  validateTokenBudget(input.tokenBudget);
412
432
  return await this.#withAccounting(async () => {
413
433
  const existing = this.#host.getState();
@@ -425,6 +445,7 @@ export class GoalRuntime {
425
445
  async replaceGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
426
446
  const objective = input.objective.trim();
427
447
  if (!objective) throw new Error("objective is required when op=replace");
448
+ validateObjective(objective);
428
449
  validateTokenBudget(input.tokenBudget);
429
450
  return await this.#withAccounting(async () => {
430
451
  const existing = this.#host.getState();
package/src/tool.ts CHANGED
@@ -127,6 +127,11 @@ export interface GoalToolDeps {
127
127
  /** Independent completion/impossibility evaluator (src/evaluator.ts).
128
128
  * Gates `complete` and adjudicates `impossible`. */
129
129
  runEvaluator: GoalEvaluatorFn;
130
+ /** Called after the tool ACTIVATES a goal mid-run (create/resume). The
131
+ * before_agent_start injection only fires on the next agent run, so
132
+ * without this hook the run that created the goal never sees the goal
133
+ * context prompt at all (observed live in print mode). */
134
+ onActivated?: () => Promise<void>;
130
135
  }
131
136
 
132
137
  function describeEvaluatorVerdict(verdict: string): string {
@@ -172,12 +177,14 @@ export function createGoalTool(deps: GoalToolDeps): ToolDefinition<typeof GoalPa
172
177
  if (params.op === "create") {
173
178
  const created = await runtime.createGoal(validateCreateParams(params));
174
179
  response = buildGoalToolResponse(created.goal);
180
+ await deps.onActivated?.();
175
181
  } else if (params.op === "get") {
176
182
  const state = deps.getState();
177
183
  response = buildGoalToolResponse(state?.goal ?? null);
178
184
  } else if (params.op === "resume") {
179
185
  const resumed = await runtime.resumeGoal();
180
186
  response = buildGoalToolResponse(resumed.goal);
187
+ await deps.onActivated?.();
181
188
  } else if (params.op === "drop") {
182
189
  const dropped = await runtime.dropGoal();
183
190
  response = buildGoalToolResponse(dropped ?? null);
@@ -15,7 +15,6 @@ import {
15
15
  runGoalEvaluator,
16
16
  type EvaluatorSpawn,
17
17
  } from "../src/evaluator.ts";
18
-
19
18
  function spawnReturning(stdout: string): { spawn: EvaluatorSpawn; calls: { args: string[]; cwd: string }[] } {
20
19
  const calls: { args: string[]; cwd: string }[] = [];
21
20
  return {
@@ -31,6 +30,38 @@ const COMPLETE_REQUEST = { mode: "complete" as const, objective: "Ship it", clai
31
30
  const IMPOSSIBLE_REQUEST = { mode: "impossible" as const, objective: "Ship it", claim: "no network access" };
32
31
  const RUN_OPTS = { cwd: "/tmp/repo" };
33
32
 
33
+ describe("evaluator subprocess invocation", () => {
34
+ it("runs lean: no extensions, no skills (a nested goal extension would pollute the judge)", async () => {
35
+ const run = spawnReturning('{"ok": true}');
36
+ await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: run.spawn });
37
+ const { args } = run.calls[0]!;
38
+ expect(args).toContain("--no-extensions");
39
+ expect(args).toContain("--no-skills");
40
+ // Prompt stays last (argv-safety caps rely on it).
41
+ expect(args.indexOf("--no-skills")).toBeLessThan(args.length - 1);
42
+ });
43
+
44
+ it("honors GOAL_EVALUATOR_TIMEOUT_MS over the default", async () => {
45
+ const run = spawnReturning('{"ok": true}');
46
+ const timeouts: number[] = [];
47
+ const previous = process.env.GOAL_EVALUATOR_TIMEOUT_MS;
48
+ process.env.GOAL_EVALUATOR_TIMEOUT_MS = "12345";
49
+ try {
50
+ await runGoalEvaluator(COMPLETE_REQUEST, {
51
+ ...RUN_OPTS,
52
+ spawn: async (invocation, opts) => {
53
+ timeouts.push(opts.timeoutMs);
54
+ return run.spawn(invocation, opts);
55
+ },
56
+ });
57
+ } finally {
58
+ if (previous === undefined) delete process.env.GOAL_EVALUATOR_TIMEOUT_MS;
59
+ else process.env.GOAL_EVALUATOR_TIMEOUT_MS = previous;
60
+ }
61
+ expect(timeouts).toEqual([12_345]);
62
+ });
63
+ });
64
+
34
65
  describe("extractJsonObject", () => {
35
66
  it("parses a clean JSON object", () => {
36
67
  expect(extractJsonObject('{"ok": true, "reason": "tests pass"}')).toEqual({
@@ -36,6 +36,7 @@ interface FakeHost {
36
36
  customType: string;
37
37
  content: string;
38
38
  display: boolean;
39
+ details?: unknown;
39
40
  options?: { deliverAs?: string; triggerTurn?: boolean };
40
41
  }>;
41
42
  sentUserMessages: Array<{ content: string; options?: { deliverAs?: string } }>;
@@ -114,7 +115,10 @@ function fakeHost(): FakeHost {
114
115
  return host;
115
116
  }
116
117
 
117
- function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean } = {}) {
118
+ function createContext(
119
+ host: FakeHost,
120
+ overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean; mode?: string } = {},
121
+ ) {
118
122
  return {
119
123
  ui: {
120
124
  notify: (text: string) => {
@@ -130,7 +134,7 @@ function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending
130
134
  editor: async () => undefined,
131
135
  },
132
136
  hasUI: true,
133
- mode: "tui",
137
+ mode: overrides.mode ?? "tui",
134
138
  cwd: "/tmp",
135
139
  isIdle: () => overrides.idle ?? true,
136
140
  hasPendingMessages: () => overrides.pending ?? false,
@@ -148,6 +152,19 @@ async function fire(host: FakeHost, event: string, payload: unknown, ctx?: unkno
148
152
  }
149
153
  }
150
154
 
155
+ /** Fire agent_before_settle and return its boundary result. */
156
+ function fireBeforeSettle(
157
+ host: FakeHost,
158
+ ctx?: unknown,
159
+ ): { continue?: boolean; entries?: Array<{ type: string; customType?: string; details?: unknown }> } | undefined {
160
+ const list = host.handlers.get("agent_before_settle") ?? [];
161
+ let result: unknown;
162
+ for (const handler of list) {
163
+ result = handler({ type: "agent_before_settle" }, ctx ?? createContext(host));
164
+ }
165
+ return result as { continue?: boolean; entries?: Array<{ type: string; customType?: string; details?: unknown }> } | undefined;
166
+ }
167
+
151
168
  function fireBus(host: FakeHost, channel: string, data: unknown): void {
152
169
  for (const handler of host.busHandlers.get(channel) ?? []) {
153
170
  handler(data);
@@ -222,6 +239,14 @@ describe("pi-goal extension wiring", () => {
222
239
  expect(host.activeTools).toContain("read");
223
240
  });
224
241
 
242
+ it("session_start keeps the goal tool in non-interactive modes (no slash commands there)", async () => {
243
+ const host = fakeHost();
244
+ defaultExport(host.pi);
245
+ const ctx = createContext(host, { mode: "print" });
246
+ await fire(host, "session_start", { type: "session_start", reason: "startup" }, ctx);
247
+ expect(host.activeTools).toContain("goal");
248
+ });
249
+
225
250
  it("session_start restores a persisted active goal, re-adds the tool, then pauses it (omp onThreadResumed)", async () => {
226
251
  const host = fakeHost();
227
252
  defaultExport(host.pi);
@@ -256,30 +281,52 @@ describe("pi-goal extension wiring", () => {
256
281
  expect(host.statuses.get("goal")).toBe("🎯 Goal 0");
257
282
  });
258
283
 
259
- it("before_agent_start injects the hidden goal-mode-context message only while a goal is active", async () => {
284
+ it("context_with_system injects the goal-mode-context message only while a goal is active", async () => {
260
285
  const host = fakeHost();
261
286
  defaultExport(host.pi);
262
287
  await fire(host, "session_start", { type: "session_start", reason: "startup" });
263
288
 
264
- let result: unknown;
265
- const handler = host.handlers.get("before_agent_start")?.[0];
266
- result = await handler!({ prompt: "hi", systemPrompt: "" }, createContext(host));
289
+ const handler = host.handlers.get("context_with_system")?.[0]!;
290
+ const transcript = [{ role: "user", content: "hi" }];
291
+ let result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as
292
+ | { messages: Array<{ customType?: string; content?: string }> }
293
+ | undefined;
267
294
  expect(result).toBeUndefined();
268
295
 
269
296
  await host.commands.goal!.handler("Do the thing", createContext(host));
270
- result = await handler!({ prompt: "hi", systemPrompt: "" }, createContext(host));
271
- expect(result).toMatchObject({ message: { customType: "goal-mode-context", display: false } });
272
- const content = (result as { message: { content: string } }).message.content;
273
- expect(content).toContain("<goal_context>");
274
- expect(content).toContain("Do the thing");
297
+ result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as
298
+ | { messages: Array<{ customType?: string; content?: string }> }
299
+ | undefined;
300
+ const contexts = result!.messages.filter((m) => m.customType === "goal-mode-context");
301
+ expect(contexts).toHaveLength(1);
302
+ expect(contexts[0]?.content).toContain("<goal_context>");
303
+ expect(contexts[0]?.content).toContain("Do the thing");
275
304
  });
276
305
 
277
- it("agent_end schedules a continuation when the goal is still active", async () => {
306
+ it("context_with_system supersedes steer-injected transcript copies (single fresh context)", async () => {
307
+ const host = fakeHost();
308
+ defaultExport(host.pi);
309
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
310
+ await host.commands.goal!.handler("Do the thing", createContext(host));
311
+
312
+ const handler = host.handlers.get("context_with_system")?.[0]!;
313
+ const transcript = [
314
+ { role: "user", content: "hi" },
315
+ { role: "custom", customType: "goal-mode-context", content: "stale copy", display: false },
316
+ ];
317
+ const result = (await handler({ type: "context_with_system", messages: transcript }, createContext(host))) as {
318
+ messages: Array<{ customType?: string; content?: string }>;
319
+ };
320
+ const contexts = result.messages.filter((m) => m.customType === "goal-mode-context");
321
+ expect(contexts).toHaveLength(1);
322
+ expect(contexts[0]?.content).not.toBe("stale copy");
323
+ });
324
+
325
+ it("agent_before_settle continues exactly one request when the goal is still active", async () => {
278
326
  const host = fakeHost();
279
327
  defaultExport(host.pi);
280
328
  await fire(host, "session_start", { type: "session_start", reason: "startup" });
281
329
  await host.commands.goal!.handler("Do the thing", createContext(host));
282
- host.sentMessages.length = 0;
283
330
 
284
331
  await fire(host, "agent_start", { type: "agent_start" });
285
332
  await fire(host, "tool_execution_end", {
@@ -291,10 +338,16 @@ describe("pi-goal extension wiring", () => {
291
338
  });
292
339
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("toolUse")] });
293
340
 
294
- const continuation = host.sentMessages.find((m) => m.customType === "goal-continuation");
295
- expect(continuation).toBeDefined();
296
- expect(continuation?.display).toBe(false);
297
- expect(continuation?.options).toMatchObject({ triggerTurn: true, deliverAs: "followUp" });
341
+ const result = fireBeforeSettle(host);
342
+ expect(result?.continue).toBe(true);
343
+ // The continuation prompt rides as a custom_message draft.
344
+ const draft = result?.entries?.[0];
345
+ expect(draft).toMatchObject({ type: "custom_message", customType: "goal-continuation", display: false });
346
+ // details.goalId keys the context-pruning handler.
347
+ const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
348
+ expect(draft?.details).toMatchObject({ goalId });
349
+ // The boundary path sends no followUp message.
350
+ expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
298
351
  });
299
352
 
300
353
  it("suppresses the next continuation when a continuation turn produced no tool calls", async () => {
@@ -302,18 +355,16 @@ describe("pi-goal extension wiring", () => {
302
355
  defaultExport(host.pi);
303
356
  await fire(host, "session_start", { type: "session_start", reason: "startup" });
304
357
  await host.commands.goal!.handler("Do the thing", createContext(host));
305
- host.sentMessages.length = 0;
306
358
 
307
359
  // Continuation turn: agent replies with no tool calls.
308
360
  await fire(host, "agent_start", { type: "agent_start" });
309
361
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
310
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(1);
362
+ expect(fireBeforeSettle(host)?.continue).toBe(true);
311
363
 
312
- // That turn's end marks suppression: the following end schedules nothing.
313
- host.sentMessages.length = 0;
364
+ // That turn's end marks suppression: the following settle continues nothing.
314
365
  await fire(host, "agent_start", { type: "agent_start" });
315
366
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
316
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
367
+ expect(fireBeforeSettle(host)?.continue).toBeUndefined();
317
368
 
318
369
  // A real user message re-arms the loop.
319
370
  await fire(host, "message_start", { type: "message_start", message: { role: "user" } });
@@ -325,9 +376,8 @@ describe("pi-goal extension wiring", () => {
325
376
  result: {},
326
377
  isError: false,
327
378
  });
328
- host.sentMessages.length = 0;
329
379
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
330
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(1);
380
+ expect(fireBeforeSettle(host)?.continue).toBe(true);
331
381
  });
332
382
 
333
383
  it("interrupt aborts pause the goal instead of continuing", async () => {
@@ -340,7 +390,7 @@ describe("pi-goal extension wiring", () => {
340
390
  await fire(host, "agent_start", { type: "agent_start" });
341
391
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("aborted")] });
342
392
 
343
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
393
+ expect(fireBeforeSettle(host)?.continue).toBeUndefined();
344
394
  // omp footer segment: pause icon + usage.
345
395
  expect(host.statuses.get("goal")).toBe("⏸ Goal 0");
346
396
  const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
@@ -365,7 +415,7 @@ describe("pi-goal extension wiring", () => {
365
415
  expect(host.entries.some((e) => e.customType === GOAL_CLEARED_ENTRY_TYPE)).toBe(true);
366
416
  expect(host.notifications).toContain("Goal mode completed.");
367
417
  expect(host.statuses.has("goal")).toBe(false);
368
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
418
+ expect(fireBeforeSettle(host)?.continue).toBeUndefined();
369
419
  });
370
420
 
371
421
  it("budget-limit steering sends one hidden steer message when usage crosses the budget", async () => {
@@ -428,6 +478,96 @@ describe("pi-goal extension wiring", () => {
428
478
  expect(host.sentMessages.filter((m) => m.customType === "goal-budget-limit")).toHaveLength(0);
429
479
  });
430
480
 
481
+ it("budget flip attaches context-slimming drafts when PI_GOAL_SLIM_ON_BUDGET=1", async () => {
482
+ const host = fakeHost();
483
+ defaultExport(host.pi);
484
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
485
+ await host.commands.goal!.handler("Do the thing", createContext(host));
486
+ await host.commands.goal!.handler("budget 10", createContext(host));
487
+
488
+ const entries: unknown[] = [
489
+ { type: "message", id: "t-old", message: { role: "toolResult" } },
490
+ { type: "message", id: "u1", message: { role: "user" } },
491
+ { type: "message", id: "t-new", message: { role: "toolResult" } },
492
+ ];
493
+ const ctx = createContext(host, { entries });
494
+
495
+ vi.stubEnv("PI_GOAL_SLIM_ON_BUDGET", "1");
496
+ await fire(host, "turn_start", { type: "turn_start", turnIndex: 0, timestamp: 0 }, ctx);
497
+ entries.push({
498
+ type: "message",
499
+ id: "m1",
500
+ message: { role: "assistant", stopReason: "toolUse", usage: { input: 25, output: 0, cacheRead: 0, cacheWrite: 0 } },
501
+ });
502
+ await fire(host, "tool_execution_end", {
503
+ type: "tool_execution_end",
504
+ toolCallId: "t1",
505
+ toolName: "read",
506
+ result: {},
507
+ isError: false,
508
+ }, ctx);
509
+
510
+ // Budget flip arms slimming; the boundary attaches the drafts.
511
+ // Budget-limited goals stop auto-continuing (omp semantics), so the
512
+ // boundary persists the edits without a next request.
513
+ const result = fireBeforeSettle(host, ctx);
514
+ expect(result?.continue).toBeUndefined();
515
+ const edits = (result?.entries ?? []).filter((e) => e.type === "context_edit");
516
+ expect(edits).toEqual([{ type: "context_edit", targetId: "t-old", replacement: null }]);
517
+
518
+ // Once per goal: the next settle carries no further slimming drafts.
519
+ entries.push({
520
+ type: "message",
521
+ id: "m2",
522
+ message: { role: "assistant", stopReason: "toolUse", usage: { input: 50, output: 0, cacheRead: 0, cacheWrite: 0 } },
523
+ });
524
+ await fire(host, "tool_execution_end", {
525
+ type: "tool_execution_end",
526
+ toolCallId: "t2",
527
+ toolName: "read",
528
+ result: {},
529
+ isError: false,
530
+ }, ctx);
531
+ expect(fireBeforeSettle(host, ctx)).toBeUndefined();
532
+ vi.unstubAllEnvs();
533
+ });
534
+
535
+ it("context slimming stays off by default (opt-in via PI_GOAL_SLIM_ON_BUDGET)", async () => {
536
+ const host = fakeHost();
537
+ defaultExport(host.pi);
538
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
539
+ await host.commands.goal!.handler("Do the thing", createContext(host));
540
+ await host.commands.goal!.handler("budget 10", createContext(host));
541
+
542
+ const entries: unknown[] = [
543
+ { type: "message", id: "t-old", message: { role: "toolResult" } },
544
+ { type: "message", id: "u1", message: { role: "user" } },
545
+ { type: "message", id: "t-new", message: { role: "toolResult" } },
546
+ ];
547
+ const ctx = createContext(host, { entries });
548
+
549
+ // Explicitly opt out (order-independent against leaking stubs).
550
+ vi.stubEnv("PI_GOAL_SLIM_ON_BUDGET", "");
551
+ await fire(host, "turn_start", { type: "turn_start", turnIndex: 0, timestamp: 0 }, ctx);
552
+ entries.push({
553
+ type: "message",
554
+ id: "m1",
555
+ message: { role: "assistant", stopReason: "toolUse", usage: { input: 25, output: 0, cacheRead: 0, cacheWrite: 0 } },
556
+ });
557
+ await fire(host, "tool_execution_end", {
558
+ type: "tool_execution_end",
559
+ toolCallId: "t1",
560
+ toolName: "read",
561
+ result: {},
562
+ isError: false,
563
+ }, ctx);
564
+
565
+ // Budget flip steers but arms no slimming drafts; budget-limited goals
566
+ // do not auto-continue, so the boundary settles empty.
567
+ expect(fireBeforeSettle(host, ctx)).toBeUndefined();
568
+ vi.unstubAllEnvs();
569
+ });
570
+
431
571
  it("/guided-goal queues a hidden interview kickoff", async () => {
432
572
  const host = fakeHost();
433
573
  defaultExport(host.pi);
@@ -488,7 +628,8 @@ describe("pi-goal extension wiring", () => {
488
628
 
489
629
  await fire(host, "agent_start", { type: "agent_start" });
490
630
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
491
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
631
+ // Modal open at settle: the boundary withholds the continuation.
632
+ expect(fireBeforeSettle(host)?.continue).toBeUndefined();
492
633
 
493
634
  // Dialog closes while idle: the withheld continuation is scheduled now.
494
635
  const idleCtx = createContext(host, { idle: true });
@@ -517,7 +658,8 @@ describe("pi-goal extension wiring", () => {
517
658
  expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
518
659
 
519
660
  await fire(host, "agent_end", { type: "agent_end", messages: [assistantMessage("stop")] });
520
- expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(1);
661
+ // Modal already closed: the boundary itself continues the run.
662
+ expect(fireBeforeSettle(host)?.continue).toBe(true);
521
663
  });
522
664
 
523
665
  it("refreshes the status on agent_settled (status integrations per docs)", async () => {
@@ -632,4 +774,77 @@ describe("pi-goal extension wiring", () => {
632
774
  expect(component.text).toContain("6K (no budget) tokens");
633
775
  expect(component.text).toContain("20s");
634
776
  });
777
+
778
+ it("pauses the goal instead of continuing when the run ends with a provider error", async () => {
779
+ const host = fakeHost();
780
+ defaultExport(host.pi);
781
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
782
+ await host.commands.goal!.handler("Do the thing", createContext(host));
783
+ host.sentMessages.length = 0;
784
+
785
+ await fire(host, "agent_start", { type: "agent_start" });
786
+ await fire(host, "agent_end", {
787
+ type: "agent_end",
788
+ messages: [
789
+ { role: "assistant", stopReason: "error", errorMessage: "429 rate limit exceeded", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
790
+ ],
791
+ });
792
+
793
+ // Paused and persisted, with a classified notice; no continuation queued.
794
+ const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
795
+ expect(pauseEntry?.data).toMatchObject({ enabled: false, goal: { status: "paused" } });
796
+ expect(host.notifications.some((n) => n.includes("rate limits"))).toBe(true);
797
+ expect(fireBeforeSettle(host)?.continue).toBeUndefined();
798
+ });
799
+
800
+ it("pauses with a generic notice on non-usage errors", async () => {
801
+ const host = fakeHost();
802
+ defaultExport(host.pi);
803
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
804
+ await host.commands.goal!.handler("Do the thing", createContext(host));
805
+
806
+ await fire(host, "agent_start", { type: "agent_start" });
807
+ await fire(host, "agent_end", {
808
+ type: "agent_end",
809
+ messages: [
810
+ { role: "assistant", stopReason: "error", errorMessage: "socket hang up", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
811
+ ],
812
+ });
813
+
814
+ expect(host.notifications.some((n) => n.includes("ended with an error"))).toBe(true);
815
+ });
816
+
817
+ it("context pruning keeps only the newest goal messages and drops stale continuations", async () => {
818
+ const host = fakeHost();
819
+ defaultExport(host.pi);
820
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
821
+ await host.commands.goal!.handler("Do the thing", createContext(host));
822
+ const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
823
+
824
+ const custom = (customType: string, details?: unknown) => ({ role: "custom", customType, display: false, details });
825
+ const messages = [
826
+ { role: "user", content: "hi" },
827
+ custom("goal-mode-context"), // stale context
828
+ custom("goal-continuation", { goalId: "old-goal" }), // stale goal id
829
+ custom("goal-budget-limit"),
830
+ custom("goal-mode-context"), // newest context: kept
831
+ custom("goal-continuation", { goalId }), // newest for the active goal: kept
832
+ { role: "assistant", content: "working" },
833
+ custom("goal-continuation", { goalId: "old-goal" }), // another stale: dropped
834
+ ];
835
+
836
+ const handler = host.handlers.get("context")?.[0]!;
837
+ const result = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
838
+ const kept = result.messages.filter((m) => (m as { role?: string }).role === "custom");
839
+ expect(kept.map((m) => (m as { customType: string }).customType)).toEqual([
840
+ "goal-budget-limit",
841
+ "goal-mode-context",
842
+ "goal-continuation",
843
+ ]);
844
+
845
+ // Once no goal is active, every continuation is dropped.
846
+ await host.commands.goal!.handler("drop", createContext(host));
847
+ const result2 = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
848
+ expect(result2.messages.filter((m) => (m as { customType?: string }).customType === "goal-continuation")).toHaveLength(0);
849
+ });
635
850
  });
@@ -468,4 +468,22 @@ describe("goal runtime", () => {
468
468
  const second = await harness.runtime.createGoal({ objective: "Two" });
469
469
  expect(first.goal.id).not.toBe(second.goal.id);
470
470
  });
471
+
472
+ it("rejects objectives beyond the character cap on create and replace", async () => {
473
+ const harness = createHarness();
474
+ const oversized = "x".repeat(4_001);
475
+ await expect(harness.runtime.createGoal({ objective: oversized })).rejects.toThrow(/too long.*4,000/s);
476
+ // Boundary: exactly at the cap is fine.
477
+ await harness.runtime.createGoal({ objective: "x".repeat(4_000) });
478
+ // Replace path enforces the same cap.
479
+ await expect(harness.runtime.replaceGoal({ objective: oversized })).rejects.toThrow(/too long/);
480
+ });
481
+
482
+ it("counts characters as code points, not UTF-16 units", async () => {
483
+ const harness = createHarness();
484
+ // Astral emoji: 1 code point, 2 UTF-16 units. 2_001 emoji is 4_002 UTF-16
485
+ // units (a naive .length cap would reject it) but 2_001 code points: kept.
486
+ await expect(harness.runtime.createGoal({ objective: "😀".repeat(2_001) })).resolves.toBeDefined();
487
+ await expect(harness.runtime.createGoal({ objective: "😀".repeat(4_001) })).rejects.toThrow(/too long/);
488
+ });
471
489
  });
package/test/tool.test.ts CHANGED
@@ -90,8 +90,14 @@ function stubEvaluator(outcome: GoalEvaluatorOutcome) {
90
90
  function createTestTool(
91
91
  harness: ReturnType<typeof createRuntimeHarness>,
92
92
  evaluator: (request: GoalEvaluatorRequest, opts: { cwd?: string }) => Promise<GoalEvaluatorOutcome>,
93
+ onActivated?: () => Promise<void>,
93
94
  ) {
94
- return createGoalTool({ getRuntime: () => harness.runtime, getState: harness.getState, runEvaluator: evaluator });
95
+ return createGoalTool({
96
+ getRuntime: () => harness.runtime,
97
+ getState: harness.getState,
98
+ runEvaluator: evaluator,
99
+ onActivated,
100
+ });
95
101
  }
96
102
 
97
103
  // ---------------------------------------------------------------------------
@@ -99,6 +105,25 @@ function createTestTool(
99
105
  // ---------------------------------------------------------------------------
100
106
 
101
107
  describe("goal tool", () => {
108
+ it("onActivated hook fires on create and resume (mid-run context injection)", async () => {
109
+ const harness = createRuntimeHarness();
110
+ const activations: string[] = [];
111
+ const tool = createTestTool(
112
+ harness,
113
+ async () => ({ status: "confirmed", reason: "ok" }),
114
+ async () => {
115
+ activations.push("activated");
116
+ },
117
+ );
118
+ await executeTool(tool, { op: "create", objective: "Ship it" });
119
+ expect(activations).toHaveLength(1);
120
+ await executeTool(tool, { op: "resume" });
121
+ expect(activations).toHaveLength(2);
122
+ // get does not re-activate.
123
+ await executeTool(tool, { op: "get" });
124
+ expect(activations).toHaveLength(2);
125
+ });
126
+
102
127
  it("create starts a goal and returns the objective/status text", async () => {
103
128
  const harness = createRuntimeHarness();
104
129
  const tool = createTestTool(harness, async () => ({ status: "confirmed", reason: "verified" }));