@fyeeme/pi-goal 1.0.2 → 1.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,21 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ### Added
6
+
7
+ - Objective length cap (`MAX_OBJECTIVE_CHARS`, 4,000 code points) enforced by the runtime on `create` and `replace` — the objective is re-injected into context on every continuation and passed to the evaluator subprocess, so an oversized one silently burns budget every turn; the error guides long instructions into a referenced file. Counting is code-point-based, not UTF-16 units. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
8
+
9
+ ### Fixed
10
+
11
+ - Print/json/rpc modes could never create a goal: `session_start` removed the `goal` tool from the active set when no goal existed (omp sdk.ts parity), but non-interactive modes have no `/goal` or `/guided-goal` command to re-arm it — leaving the model unable to start a goal at all (found via live goal-mode testing). The tool is now only removed in TUI mode.
12
+ - The run that creates/resumes a goal via the tool never saw the goal context prompt: `before_agent_start` injection only fires on the next agent run, and the command-path steer was not wired to the tool path (found via live goal-mode testing). The tool now fires an `onActivated` hook so the host injects the goal context into the current run as a steer.
13
+ - The evaluator subprocess ran with full extension discovery: any other installed goal extension (with an active-goal system prompt) would pollute the judge that is supposed to be independent, and the startup overhead contributed to live 300s timeouts. It now runs lean (`--no-extensions --no-skills`), the default timeout is raised to 600s, and `GOAL_EVALUATOR_TIMEOUT_MS` overrides it.
14
+
15
+ ### Changed
16
+
17
+ - Context hygiene: hidden goal messages no longer accumulate in the LLM's view. A `context` handler keeps only the newest `goal-mode-context` and `goal-budget-limit` messages plus the newest `goal-continuation` stamped for the currently active goal id (`details.goalId`); stale ones — including all continuations once no goal is active — are dropped from the model's view, not from the transcript. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
18
+ - Run-error handling: when a run ends with an assistant `stopReason: "error"`, the active goal now pauses (persisted) instead of letting the continuation loop fire into a likely retry loop, with a classified notice — provider usage/rate/quota/limit errors read differently from generic faults. Abort behavior is unchanged. (Borrowed from mitsuhiko/agent-stuff `extensions/goal.ts`.)
19
+
5
20
  ## [1.0.2] - 2026-09-16
6
21
 
7
22
  ### Changed
package/index.ts CHANGED
@@ -97,6 +97,7 @@ const goalModeContextPrompt = readFileSync(
97
97
  interface EntryMessageLike {
98
98
  role?: string;
99
99
  stopReason?: string;
100
+ errorMessage?: string;
100
101
  usage?: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined;
101
102
  }
102
103
 
@@ -340,6 +341,23 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
340
341
  await exitGoalMode({ reason: "paused" });
341
342
  }
342
343
 
344
+ /** A provider/agent error ended the run (borrowed from mitsuhiko/agent-stuff
345
+ * goal.ts): pause the active goal instead of letting scheduleContinuation
346
+ * fire into a likely retry loop, and classify the failure for the user.
347
+ * Usage/rate/quota errors read differently from generic faults. */
348
+ async function pauseOnRunError(messages: unknown[]): Promise<void> {
349
+ if (!goalState?.enabled || goalState.goal.status !== "active") return;
350
+ const errorMessage = lastAssistantErrorMessage(messages) ?? "";
351
+ const usageLimited = /\b(usage|rate|quota|limit)\b/i.test(errorMessage);
352
+ await runtime.pauseGoal();
353
+ notify(
354
+ usageLimited
355
+ ? "Goal paused: the last turn hit provider usage/rate limits. /goal resume when ready."
356
+ : "Goal paused: the last turn ended with an error. /goal resume to continue.",
357
+ "error",
358
+ );
359
+ }
360
+
343
361
  async function dropGoal(): Promise<void> {
344
362
  if (!goalState) {
345
363
  notify("No goal to drop.", "warning");
@@ -408,7 +426,9 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
408
426
  if (!prompt) return;
409
427
  continuationInFlight = true;
410
428
  pi.sendMessage(
411
- { customType: "goal-continuation", content: prompt, display: false },
429
+ // details.goalId keys context pruning (see the "context" handler below):
430
+ // only the newest continuation of the CURRENTLY active goal survives.
431
+ { customType: "goal-continuation", content: prompt, display: false, details: { goalId: goalState.goal.id } },
412
432
  { triggerTurn: true, deliverAs: "followUp" },
413
433
  );
414
434
  }
@@ -425,6 +445,11 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
425
445
  // fall back to the process cwd.
426
446
  runEvaluator: (request, opts) =>
427
447
  runGoalEvaluator(request, { cwd: opts.cwd ?? currentCtx?.cwd ?? process.cwd(), signal: opts.signal }),
448
+ // Mid-run activation (tool create/resume): inject the goal context into
449
+ // the CURRENT run as a steer — before_agent_start only covers the next run.
450
+ onActivated: async () => {
451
+ if (currentCtx && !currentCtx.isIdle()) await sendGoalModeContext("steer");
452
+ },
428
453
  };
429
454
  pi.registerTool(createGoalTool(goalToolDeps));
430
455
 
@@ -501,11 +526,16 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
501
526
  const restored = restoreGoalFromEntries(branch);
502
527
  if (!restored) {
503
528
  goalState = undefined;
504
- // omp sdk.ts excludes the goal tool from the initial set; mirror that
505
- // when this session has no goal to manage.
506
- const active = pi.getActiveTools();
507
- if (active.includes("goal")) {
508
- pi.setActiveTools(active.filter((name) => name !== "goal"));
529
+ // omp sdk.ts excludes the goal tool from the initial set; mirror that in
530
+ // TUI where /goal and /guided-goal re-arm it. In non-interactive modes
531
+ // (print/json/rpc) no slash command exists, so removing the tool would
532
+ // leave the model unable to create a goal at all (found via live
533
+ // goal-mode testing): keep it active there.
534
+ if (ctx.mode === "tui") {
535
+ const active = pi.getActiveTools();
536
+ if (active.includes("goal")) {
537
+ pi.setActiveTools(active.filter((name) => name !== "goal"));
538
+ }
509
539
  }
510
540
  updateStatus();
511
541
  return;
@@ -563,15 +593,51 @@ export default function piGoalExtension(pi: ExtensionAPI): void {
563
593
  };
564
594
  });
565
595
 
596
+ // Context hygiene (borrowed from mitsuhiko/agent-stuff goal.ts): hidden goal
597
+ // messages otherwise accumulate one per run/continuation and burn context
598
+ // every LLM call. Keep only the newest goal-mode-context and
599
+ // goal-budget-limit, plus the newest continuation stamped for the currently
600
+ // active goal id; stale ones (including all continuations once no goal is
601
+ // active) are dropped from the model's view, not from the transcript.
602
+ pi.on("context", (event) => {
603
+ const activeGoalId = goalState?.enabled && goalState.goal.status === "active" ? goalState.goal.id : undefined;
604
+ let lastContext = -1;
605
+ let lastBudget = -1;
606
+ let lastContinuation = -1;
607
+ for (let i = 0; i < event.messages.length; i++) {
608
+ const msg = event.messages[i] as
609
+ | { role?: string; customType?: string; details?: { goalId?: string } }
610
+ | undefined;
611
+ if (msg?.role !== "custom") continue;
612
+ if (msg.customType === "goal-mode-context") lastContext = i;
613
+ else if (msg.customType === "goal-budget-limit") lastBudget = i;
614
+ else if (msg.customType === "goal-continuation" && activeGoalId !== undefined && msg.details?.goalId === activeGoalId) {
615
+ lastContinuation = i;
616
+ }
617
+ }
618
+ if (lastContext === -1 && lastBudget === -1 && lastContinuation === -1) return undefined;
619
+ return {
620
+ messages: event.messages.filter((message, index) => {
621
+ const msg = message as { role?: string; customType?: string } | undefined;
622
+ if (msg?.role !== "custom") return true;
623
+ if (msg.customType === "goal-mode-context") return index === lastContext;
624
+ if (msg.customType === "goal-budget-limit") return index === lastBudget;
625
+ if (msg.customType === "goal-continuation") return index === lastContinuation;
626
+ return true;
627
+ }),
628
+ };
629
+ });
630
+
566
631
  pi.on("agent_end", async (event, ctx) => {
567
632
  currentCtx = ctx;
568
633
  // omp separates onAgentEnd (session) from continuation scheduling
569
634
  // (interactive-mode); both subscribe to the same end-of-run moment.
570
- const aborted = lastAssistantStopReason(event.messages) === "aborted";
571
- if (aborted) {
635
+ const stopReason = lastAssistantStopReason(event.messages);
636
+ if (stopReason === "aborted") {
572
637
  await runtime.onTaskAborted({ reason: "interrupted" });
573
638
  } else {
574
639
  await runtime.onAgentEnd({ currentUsage: currentUsage(ctx) });
640
+ if (stopReason === "error") await pauseOnRunError(event.messages);
575
641
  }
576
642
 
577
643
  if (continuationInFlight) {
@@ -665,6 +731,14 @@ function lastAssistantStopReason(messages: unknown[]): string | undefined {
665
731
  return undefined;
666
732
  }
667
733
 
734
+ function lastAssistantErrorMessage(messages: unknown[]): string | undefined {
735
+ for (let i = messages.length - 1; i >= 0; i--) {
736
+ const message = messages[i] as EntryMessageLike | undefined;
737
+ if (message?.role === "assistant") return message.errorMessage;
738
+ }
739
+ return undefined;
740
+ }
741
+
668
742
  export type { GoalRuntimeHost } from "./src/runtime.ts";
669
743
  // Re-exported for consumers that compose the pieces directly (tests, tools).
670
744
  export { GoalRuntime } from "./src/runtime.ts";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@fyeeme/pi-goal",
3
- "version": "1.0.2",
3
+ "version": "1.0.3",
4
4
  "description": "Goal mode for pi \u2014 one persistent autonomous objective looped until verified success: goal tool (create/get/complete/resume/drop), token/time budget accounting with budget-limit steering, automatic continuation turns, /goal and /guided-goal commands, and a goal_updated event bus for other extensions.",
5
5
  "type": "module",
6
6
  "license": "MIT",
package/src/evaluator.ts CHANGED
@@ -34,8 +34,16 @@ const evaluatorCompletePrompt = readFileSync(path.join(promptsDir, "evaluator-co
34
34
  const evaluatorImpossiblePrompt = readFileSync(path.join(promptsDir, "evaluator-impossible.md"), "utf8");
35
35
 
36
36
  /** Hard cap on one evaluator run — the grounded check may run test suites,
37
- * so this is deliberately generous; anything longer has failed. */
38
- export const EVALUATOR_TIMEOUT_MS = 300_000;
37
+ * so this is deliberately generous; anything longer has failed. Each LLM
38
+ * call on slow providers can take 20s+, and the evaluator is a multi-step
39
+ * agent (inspect repo, run checks, emit JSON), so the default needs real
40
+ * headroom. Override with GOAL_EVALUATOR_TIMEOUT_MS. */
41
+ export const EVALUATOR_TIMEOUT_MS = 600_000;
42
+
43
+ function evaluatorTimeout(): number {
44
+ const raw = Number.parseInt(process.env.GOAL_EVALUATOR_TIMEOUT_MS ?? "", 10);
45
+ return Number.isInteger(raw) && raw > 0 ? raw : EVALUATOR_TIMEOUT_MS;
46
+ }
39
47
 
40
48
  /** Argv-safety caps (kernel MAX_ARG_STRLEN ≈ 128KB; these keep the prompt
41
49
  * argument far under it even for verbose objectives/audits). */
@@ -118,6 +126,16 @@ export function buildEvaluatorPrompt(request: GoalEvaluatorRequest): string {
118
126
  });
119
127
  }
120
128
 
129
+ /** Build the evaluator argv: a one-shot headless run. Lean flags matter here:
130
+ * without them the nested pi loads the user's full extension set — including
131
+ * any OTHER goal extension, whose active-goal system prompt would pollute the
132
+ * very evaluator that is supposed to judge the goal independently — and its
133
+ * startup cost pushes long verification runs over the timeout (observed
134
+ * live: 300s timeout hit while the evaluator was still working). */
135
+ function evaluatorInvocation(prompt: string): { command: string; args: string[] } {
136
+ return getPiInvocation(["-p", "--no-session", "--no-extensions", "--no-skills", prompt]);
137
+ }
138
+
121
139
  /** Extract the first balanced JSON object from evaluator output. Handles the
122
140
  * clean case, code-fenced output, and prose-wrapped JSON; anything else is
123
141
  * unparseable. Never throws. */
@@ -181,10 +199,10 @@ export async function runGoalEvaluator(
181
199
  ): Promise<GoalEvaluatorOutcome> {
182
200
  const prompt = buildEvaluatorPrompt(request);
183
201
  const spawn = opts.spawn ?? defaultSpawn;
184
- const timeoutMs = opts.timeoutMs ?? EVALUATOR_TIMEOUT_MS;
202
+ const timeoutMs = opts.timeoutMs ?? evaluatorTimeout();
185
203
  let stdout: string;
186
204
  try {
187
- stdout = await spawn(getPiInvocation(["-p", "--no-session", prompt]), {
205
+ stdout = await spawn(evaluatorInvocation(prompt), {
188
206
  cwd: opts.cwd,
189
207
  timeoutMs,
190
208
  signal: opts.signal,
package/src/runtime.ts CHANGED
@@ -128,6 +128,25 @@ export function completionBudgetReport(goal: Goal): string | null {
128
128
  return `Goal achieved. Report final budget usage to the user: ${parts.join("; ")}.`;
129
129
  }
130
130
 
131
+ /** Objective length cap. The objective is re-injected into context on every
132
+ * continuation and passed to the evaluator subprocess, so an oversized one
133
+ * silently burns budget every turn. Borrowed from mitsuhiko/agent-stuff
134
+ * goal.ts (MAX_OBJECTIVE_CHARS). */
135
+ export const MAX_OBJECTIVE_CHARS = 4_000;
136
+
137
+ function charCount(value: string): number {
138
+ return [...value].length;
139
+ }
140
+
141
+ function validateObjective(objective: string): void {
142
+ const count = charCount(objective);
143
+ if (count > MAX_OBJECTIVE_CHARS) {
144
+ throw new Error(
145
+ `Goal objective is too long: ${count.toLocaleString()} characters. Limit: ${MAX_OBJECTIVE_CHARS.toLocaleString()} characters. Put longer instructions in a file and reference that file in the objective.`,
146
+ );
147
+ }
148
+ }
149
+
131
150
  function validateTokenBudget(tokenBudget: number | undefined): void {
132
151
  if (tokenBudget !== undefined && (!Number.isInteger(tokenBudget) || tokenBudget <= 0)) {
133
152
  throw new Error("goal token_budget must be a positive integer when provided");
@@ -408,6 +427,7 @@ export class GoalRuntime {
408
427
  async createGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
409
428
  const objective = input.objective.trim();
410
429
  if (!objective) throw new Error("objective is required when op=create");
430
+ validateObjective(objective);
411
431
  validateTokenBudget(input.tokenBudget);
412
432
  return await this.#withAccounting(async () => {
413
433
  const existing = this.#host.getState();
@@ -425,6 +445,7 @@ export class GoalRuntime {
425
445
  async replaceGoal(input: { objective: string; tokenBudget?: number }): Promise<GoalModeState> {
426
446
  const objective = input.objective.trim();
427
447
  if (!objective) throw new Error("objective is required when op=replace");
448
+ validateObjective(objective);
428
449
  validateTokenBudget(input.tokenBudget);
429
450
  return await this.#withAccounting(async () => {
430
451
  const existing = this.#host.getState();
package/src/tool.ts CHANGED
@@ -127,6 +127,11 @@ export interface GoalToolDeps {
127
127
  /** Independent completion/impossibility evaluator (src/evaluator.ts).
128
128
  * Gates `complete` and adjudicates `impossible`. */
129
129
  runEvaluator: GoalEvaluatorFn;
130
+ /** Called after the tool ACTIVATES a goal mid-run (create/resume). The
131
+ * before_agent_start injection only fires on the next agent run, so
132
+ * without this hook the run that created the goal never sees the goal
133
+ * context prompt at all (observed live in print mode). */
134
+ onActivated?: () => Promise<void>;
130
135
  }
131
136
 
132
137
  function describeEvaluatorVerdict(verdict: string): string {
@@ -172,12 +177,14 @@ export function createGoalTool(deps: GoalToolDeps): ToolDefinition<typeof GoalPa
172
177
  if (params.op === "create") {
173
178
  const created = await runtime.createGoal(validateCreateParams(params));
174
179
  response = buildGoalToolResponse(created.goal);
180
+ await deps.onActivated?.();
175
181
  } else if (params.op === "get") {
176
182
  const state = deps.getState();
177
183
  response = buildGoalToolResponse(state?.goal ?? null);
178
184
  } else if (params.op === "resume") {
179
185
  const resumed = await runtime.resumeGoal();
180
186
  response = buildGoalToolResponse(resumed.goal);
187
+ await deps.onActivated?.();
181
188
  } else if (params.op === "drop") {
182
189
  const dropped = await runtime.dropGoal();
183
190
  response = buildGoalToolResponse(dropped ?? null);
@@ -15,7 +15,6 @@ import {
15
15
  runGoalEvaluator,
16
16
  type EvaluatorSpawn,
17
17
  } from "../src/evaluator.ts";
18
-
19
18
  function spawnReturning(stdout: string): { spawn: EvaluatorSpawn; calls: { args: string[]; cwd: string }[] } {
20
19
  const calls: { args: string[]; cwd: string }[] = [];
21
20
  return {
@@ -31,6 +30,38 @@ const COMPLETE_REQUEST = { mode: "complete" as const, objective: "Ship it", clai
31
30
  const IMPOSSIBLE_REQUEST = { mode: "impossible" as const, objective: "Ship it", claim: "no network access" };
32
31
  const RUN_OPTS = { cwd: "/tmp/repo" };
33
32
 
33
+ describe("evaluator subprocess invocation", () => {
34
+ it("runs lean: no extensions, no skills (a nested goal extension would pollute the judge)", async () => {
35
+ const run = spawnReturning('{"ok": true}');
36
+ await runGoalEvaluator(COMPLETE_REQUEST, { ...RUN_OPTS, spawn: run.spawn });
37
+ const { args } = run.calls[0]!;
38
+ expect(args).toContain("--no-extensions");
39
+ expect(args).toContain("--no-skills");
40
+ // Prompt stays last (argv-safety caps rely on it).
41
+ expect(args.indexOf("--no-skills")).toBeLessThan(args.length - 1);
42
+ });
43
+
44
+ it("honors GOAL_EVALUATOR_TIMEOUT_MS over the default", async () => {
45
+ const run = spawnReturning('{"ok": true}');
46
+ const timeouts: number[] = [];
47
+ const previous = process.env.GOAL_EVALUATOR_TIMEOUT_MS;
48
+ process.env.GOAL_EVALUATOR_TIMEOUT_MS = "12345";
49
+ try {
50
+ await runGoalEvaluator(COMPLETE_REQUEST, {
51
+ ...RUN_OPTS,
52
+ spawn: async (invocation, opts) => {
53
+ timeouts.push(opts.timeoutMs);
54
+ return run.spawn(invocation, opts);
55
+ },
56
+ });
57
+ } finally {
58
+ if (previous === undefined) delete process.env.GOAL_EVALUATOR_TIMEOUT_MS;
59
+ else process.env.GOAL_EVALUATOR_TIMEOUT_MS = previous;
60
+ }
61
+ expect(timeouts).toEqual([12_345]);
62
+ });
63
+ });
64
+
34
65
  describe("extractJsonObject", () => {
35
66
  it("parses a clean JSON object", () => {
36
67
  expect(extractJsonObject('{"ok": true, "reason": "tests pass"}')).toEqual({
@@ -36,6 +36,7 @@ interface FakeHost {
36
36
  customType: string;
37
37
  content: string;
38
38
  display: boolean;
39
+ details?: unknown;
39
40
  options?: { deliverAs?: string; triggerTurn?: boolean };
40
41
  }>;
41
42
  sentUserMessages: Array<{ content: string; options?: { deliverAs?: string } }>;
@@ -114,7 +115,10 @@ function fakeHost(): FakeHost {
114
115
  return host;
115
116
  }
116
117
 
117
- function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean } = {}) {
118
+ function createContext(
119
+ host: FakeHost,
120
+ overrides: { entries?: unknown[]; pending?: boolean; idle?: boolean; mode?: string } = {},
121
+ ) {
118
122
  return {
119
123
  ui: {
120
124
  notify: (text: string) => {
@@ -130,7 +134,7 @@ function createContext(host: FakeHost, overrides: { entries?: unknown[]; pending
130
134
  editor: async () => undefined,
131
135
  },
132
136
  hasUI: true,
133
- mode: "tui",
137
+ mode: overrides.mode ?? "tui",
134
138
  cwd: "/tmp",
135
139
  isIdle: () => overrides.idle ?? true,
136
140
  hasPendingMessages: () => overrides.pending ?? false,
@@ -222,6 +226,14 @@ describe("pi-goal extension wiring", () => {
222
226
  expect(host.activeTools).toContain("read");
223
227
  });
224
228
 
229
+ it("session_start keeps the goal tool in non-interactive modes (no slash commands there)", async () => {
230
+ const host = fakeHost();
231
+ defaultExport(host.pi);
232
+ const ctx = createContext(host, { mode: "print" });
233
+ await fire(host, "session_start", { type: "session_start", reason: "startup" }, ctx);
234
+ expect(host.activeTools).toContain("goal");
235
+ });
236
+
225
237
  it("session_start restores a persisted active goal, re-adds the tool, then pauses it (omp onThreadResumed)", async () => {
226
238
  const host = fakeHost();
227
239
  defaultExport(host.pi);
@@ -295,6 +307,9 @@ describe("pi-goal extension wiring", () => {
295
307
  expect(continuation).toBeDefined();
296
308
  expect(continuation?.display).toBe(false);
297
309
  expect(continuation?.options).toMatchObject({ triggerTurn: true, deliverAs: "followUp" });
310
+ // details.goalId keys the context-pruning handler.
311
+ const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
312
+ expect(continuation?.details).toMatchObject({ goalId });
298
313
  });
299
314
 
300
315
  it("suppresses the next continuation when a continuation turn produced no tool calls", async () => {
@@ -632,4 +647,77 @@ describe("pi-goal extension wiring", () => {
632
647
  expect(component.text).toContain("6K (no budget) tokens");
633
648
  expect(component.text).toContain("20s");
634
649
  });
650
+
651
+ it("pauses the goal instead of continuing when the run ends with a provider error", async () => {
652
+ const host = fakeHost();
653
+ defaultExport(host.pi);
654
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
655
+ await host.commands.goal!.handler("Do the thing", createContext(host));
656
+ host.sentMessages.length = 0;
657
+
658
+ await fire(host, "agent_start", { type: "agent_start" });
659
+ await fire(host, "agent_end", {
660
+ type: "agent_end",
661
+ messages: [
662
+ { role: "assistant", stopReason: "error", errorMessage: "429 rate limit exceeded", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
663
+ ],
664
+ });
665
+
666
+ // Paused and persisted, with a classified notice; no continuation queued.
667
+ const pauseEntry = [...host.entries].reverse().find((e) => e.customType === GOAL_STATE_ENTRY_TYPE);
668
+ expect(pauseEntry?.data).toMatchObject({ enabled: false, goal: { status: "paused" } });
669
+ expect(host.notifications.some((n) => n.includes("rate limits"))).toBe(true);
670
+ expect(host.sentMessages.filter((m) => m.customType === "goal-continuation")).toHaveLength(0);
671
+ });
672
+
673
+ it("pauses with a generic notice on non-usage errors", async () => {
674
+ const host = fakeHost();
675
+ defaultExport(host.pi);
676
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
677
+ await host.commands.goal!.handler("Do the thing", createContext(host));
678
+
679
+ await fire(host, "agent_start", { type: "agent_start" });
680
+ await fire(host, "agent_end", {
681
+ type: "agent_end",
682
+ messages: [
683
+ { role: "assistant", stopReason: "error", errorMessage: "socket hang up", usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } },
684
+ ],
685
+ });
686
+
687
+ expect(host.notifications.some((n) => n.includes("ended with an error"))).toBe(true);
688
+ });
689
+
690
+ it("context pruning keeps only the newest goal messages and drops stale continuations", async () => {
691
+ const host = fakeHost();
692
+ defaultExport(host.pi);
693
+ await fire(host, "session_start", { type: "session_start", reason: "startup" });
694
+ await host.commands.goal!.handler("Do the thing", createContext(host));
695
+ const goalId = (host.entries.find((e) => e.customType === GOAL_STATE_ENTRY_TYPE)?.data as { goal: Goal }).goal.id;
696
+
697
+ const custom = (customType: string, details?: unknown) => ({ role: "custom", customType, display: false, details });
698
+ const messages = [
699
+ { role: "user", content: "hi" },
700
+ custom("goal-mode-context"), // stale context
701
+ custom("goal-continuation", { goalId: "old-goal" }), // stale goal id
702
+ custom("goal-budget-limit"),
703
+ custom("goal-mode-context"), // newest context: kept
704
+ custom("goal-continuation", { goalId }), // newest for the active goal: kept
705
+ { role: "assistant", content: "working" },
706
+ custom("goal-continuation", { goalId: "old-goal" }), // another stale: dropped
707
+ ];
708
+
709
+ const handler = host.handlers.get("context")?.[0]!;
710
+ const result = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
711
+ const kept = result.messages.filter((m) => (m as { role?: string }).role === "custom");
712
+ expect(kept.map((m) => (m as { customType: string }).customType)).toEqual([
713
+ "goal-budget-limit",
714
+ "goal-mode-context",
715
+ "goal-continuation",
716
+ ]);
717
+
718
+ // Once no goal is active, every continuation is dropped.
719
+ await host.commands.goal!.handler("drop", createContext(host));
720
+ const result2 = (await handler({ type: "context", messages }, createContext(host))) as { messages: unknown[] };
721
+ expect(result2.messages.filter((m) => (m as { customType?: string }).customType === "goal-continuation")).toHaveLength(0);
722
+ });
635
723
  });
@@ -468,4 +468,22 @@ describe("goal runtime", () => {
468
468
  const second = await harness.runtime.createGoal({ objective: "Two" });
469
469
  expect(first.goal.id).not.toBe(second.goal.id);
470
470
  });
471
+
472
+ it("rejects objectives beyond the character cap on create and replace", async () => {
473
+ const harness = createHarness();
474
+ const oversized = "x".repeat(4_001);
475
+ await expect(harness.runtime.createGoal({ objective: oversized })).rejects.toThrow(/too long.*4,000/s);
476
+ // Boundary: exactly at the cap is fine.
477
+ await harness.runtime.createGoal({ objective: "x".repeat(4_000) });
478
+ // Replace path enforces the same cap.
479
+ await expect(harness.runtime.replaceGoal({ objective: oversized })).rejects.toThrow(/too long/);
480
+ });
481
+
482
+ it("counts characters as code points, not UTF-16 units", async () => {
483
+ const harness = createHarness();
484
+ // Astral emoji: 1 code point, 2 UTF-16 units. 2_001 emoji is 4_002 UTF-16
485
+ // units (a naive .length cap would reject it) but 2_001 code points: kept.
486
+ await expect(harness.runtime.createGoal({ objective: "😀".repeat(2_001) })).resolves.toBeDefined();
487
+ await expect(harness.runtime.createGoal({ objective: "😀".repeat(4_001) })).rejects.toThrow(/too long/);
488
+ });
471
489
  });
package/test/tool.test.ts CHANGED
@@ -90,8 +90,14 @@ function stubEvaluator(outcome: GoalEvaluatorOutcome) {
90
90
  function createTestTool(
91
91
  harness: ReturnType<typeof createRuntimeHarness>,
92
92
  evaluator: (request: GoalEvaluatorRequest, opts: { cwd?: string }) => Promise<GoalEvaluatorOutcome>,
93
+ onActivated?: () => Promise<void>,
93
94
  ) {
94
- return createGoalTool({ getRuntime: () => harness.runtime, getState: harness.getState, runEvaluator: evaluator });
95
+ return createGoalTool({
96
+ getRuntime: () => harness.runtime,
97
+ getState: harness.getState,
98
+ runEvaluator: evaluator,
99
+ onActivated,
100
+ });
95
101
  }
96
102
 
97
103
  // ---------------------------------------------------------------------------
@@ -99,6 +105,25 @@ function createTestTool(
99
105
  // ---------------------------------------------------------------------------
100
106
 
101
107
  describe("goal tool", () => {
108
+ it("onActivated hook fires on create and resume (mid-run context injection)", async () => {
109
+ const harness = createRuntimeHarness();
110
+ const activations: string[] = [];
111
+ const tool = createTestTool(
112
+ harness,
113
+ async () => ({ status: "confirmed", reason: "ok" }),
114
+ async () => {
115
+ activations.push("activated");
116
+ },
117
+ );
118
+ await executeTool(tool, { op: "create", objective: "Ship it" });
119
+ expect(activations).toHaveLength(1);
120
+ await executeTool(tool, { op: "resume" });
121
+ expect(activations).toHaveLength(2);
122
+ // get does not re-activate.
123
+ await executeTool(tool, { op: "get" });
124
+ expect(activations).toHaveLength(2);
125
+ });
126
+
102
127
  it("create starts a goal and returns the objective/status text", async () => {
103
128
  const harness = createRuntimeHarness();
104
129
  const tool = createTestTool(harness, async () => ({ status: "confirmed", reason: "verified" }));