@gajae-code/agent-core 0.4.4 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,32 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.5.0] - 2026-06-13
6
+
7
+ ### Fixed
8
+
9
+ - Fixed compaction cut-point selection when the newest retained context ends in an uncuttable tool result, so automatic compaction can keep the latest assistant/tool-result pair instead of falling back to a no-op cut.
10
+
11
+ ### Changed
12
+
13
+ - Optimization Suite v3 Lane 2 (context cost): compaction token estimates now use a shared per-entry cache (`estimateEntryTokens`) keyed by a boundary-lossless fingerprint of the exact estimator fragments, covering `estimateEntriesTokens`, the `findCutPoint` reverse walk, and pruning candidate scoring — repeated full-session estimate p95 −97%, token totals exactly equal to fresh estimates and never stale after prune mutation. Pruned bash/search/grep tool results now carry a one-line digest notice (exit code, match/file count, first error line; capped at 1.25× the generic notice cost) instead of a bare truncation notice, with savings computed from the exact notice string; reads and other tools keep the generic notice. `trimOpenAiCompactInput` is O(n) via per-item serialized lengths and a running character sum (5k-item trim −99.9%) and is now exported.
14
+ - Optimization Suite v3 Lane 3 (serialization): `cloneJson` in the append-only context now uses a typed JSON-semantic recursive clone instead of a `JSON.parse(JSON.stringify())` round-trip (−36% median on clone-heavy paths), with exact JSON.stringify byte parity including the toJSON holder-key protocol (single get, no re-dispatch on replacement values), function/symbol dropping, sparse arrays, Dates, and prototype-bearing objects; the helper is now exported.
15
+
16
+ ## [0.4.5] - 2026-06-12
17
+
18
+ ### Changed
19
+
20
+ - Made tool-output pruning staleness-aware: results superseded by a later same-target result (re-read file, re-run search) or invalidated by a later successful edit/write are pruned before merely-old ones, including inside the recency protect window. New optional `PruneConfig.staleOverridableTools` (default `["read"]`) waives protected-tool immunity for superseded results while the most recent result per target stays protected. Target identity uses collision-proof canonical JSON tuple keys.
21
+ - `PruneResult` now returns `prunedEntries` so callers whose entry source materializes copies (e.g. blob-externalized session entries) can write mutations back into their canonical store.
22
+
23
+ ### Fixed
24
+
25
+ - Preserved Cursor-native tool call rendering and execution through the agent tool-call path, including runtime tool details.
26
+
27
+ ## [0.4.4] - 2026-06-10
28
+
29
+ - Version aligned with the 0.4.4 monorepo release; no functional changes in this package.
30
+
5
31
  ## [0.4.3] - 2026-06-10
6
32
 
7
33
  ### Fixed
@@ -80,6 +80,8 @@ export interface AgentOptions {
80
80
  * Use this when abort decisions must happen before buffered events continue flowing.
81
81
  */
82
82
  onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
83
+ /** Called for non-content tool-choice incapability stream events. */
84
+ onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
83
85
  /**
84
86
  * Called when GPT-5 Harmony protocol leakage is detected and mitigated.
85
87
  */
@@ -121,3 +121,4 @@ export declare class AppendOnlyContextManager {
121
121
  invalidate(): void;
122
122
  reset(context: AgentContext, options: BuildOptions): void;
123
123
  }
124
+ export declare function cloneJson<T>(value: T): T;
@@ -68,11 +68,37 @@ export declare function effectiveReserveTokens(contextWindow: number, settings:
68
68
  export declare function shouldCompact(contextTokens: number, contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): boolean;
69
69
  export declare function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings, maxOutputTokens?: number): number;
70
70
  /**
71
- * Estimate token count for a message using cl100k_base via the native
72
- * tokenizer. This is not Anthropic's first-party tokenizer (Anthropic doesn't
73
- * publish one) but is within ~5–10% across English/code text.
71
+ * Estimate token count for a message using the native o200k tokenizer.
72
+ * Exact for o200k only; an approximation for Anthropic/other model families
73
+ * (Anthropic doesn't publish a tokenizer) within ~5–10% on English/code text.
74
+ *
75
+ * This materializes the native BPE table (~50MB RSS) on first call. Use it
76
+ * only for context-changing decisions (compaction trigger/cut points, pruning
77
+ * budgets, branch summarization, fork-context seeding, context-limit
78
+ * enforcement). For display-only totals use
79
+ * {@link estimateMessageTokensHeuristic}.
80
+ */
81
+ export declare function countMessageTokensNativeO200k(message: AgentMessage): number;
82
+ /**
83
+ * Backwards-compatible alias for {@link countMessageTokensNativeO200k}.
84
+ * Existing callers treat this as the canonical message-token estimator for
85
+ * context-changing decisions.
86
+ */
87
+ export declare const estimateTokens: typeof countMessageTokensNativeO200k;
88
+ /**
89
+ * Cheap, native-free token estimate for a message. Suitable ONLY for
90
+ * display/init surfaces (status line, /context report, HUD totals) — never
91
+ * for context-changing decisions, which must use
92
+ * {@link countMessageTokensNativeO200k}.
93
+ */
94
+ export declare function estimateMessageTokensHeuristic(message: AgentMessage): number;
95
+ /**
96
+ * Cheap, native-free token estimate for plain string fragments. Display-only
97
+ * counterpart of the native `countTokens(fragments)` aggregate.
74
98
  */
75
- export declare function estimateTokens(message: AgentMessage): number;
99
+ export declare function estimateTextTokensHeuristic(fragments: string | readonly string[]): number;
100
+ export declare function estimateEntryTokens(entry: SessionEntry): number;
101
+ export declare function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number;
76
102
  /**
77
103
  * Find the user message (or bashExecution) that starts the turn containing the given entry index.
78
104
  * Returns -1 if no turn start found before the index.
@@ -20,6 +20,12 @@ export interface ModelChangeEntry extends SessionEntryBase {
20
20
  model: string;
21
21
  /** Role: "default", "smol", "slow", etc. Undefined treated as "default" */
22
22
  role?: string;
23
+ /** Requested model before a runtime substitution/fallback, in "provider/modelId" format. */
24
+ previousModel?: string;
25
+ /** Machine-readable reason for runtime model substitution/fallback. */
26
+ reason?: string;
27
+ /** Effective thinking level when the change was recorded. */
28
+ thinkingLevel?: string | null;
23
29
  }
24
30
  export interface ServiceTierChangeEntry extends SessionEntryBase {
25
31
  type: "service_tier_change";
@@ -41,6 +41,8 @@ export interface RemoteCompactionResponse {
41
41
  export declare function shouldUseOpenAiRemoteCompaction(model: Model): boolean;
42
42
  export declare function getPreservedOpenAiRemoteCompactionData(preserveData: Record<string, unknown> | undefined): OpenAiRemoteCompactionPreserveData | undefined;
43
43
  export declare function withOpenAiRemoteCompactionPreserveData(preserveData: Record<string, unknown> | undefined, remoteCompaction: OpenAiRemoteCompactionPreserveData | undefined): Record<string, unknown> | undefined;
44
+ export declare function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number;
45
+ export declare function trimOpenAiCompactInput(input: Array<Record<string, unknown>>, contextWindow: number, instructions: string): Array<Record<string, unknown>>;
44
46
  /**
45
47
  * Build the OpenAI Responses-API native history array from LLM messages.
46
48
  *
@@ -1,7 +1,13 @@
1
1
  /**
2
2
  * Tool output pruning utilities for compaction.
3
+ *
4
+ * Candidate selection is staleness-aware: tool results that have been
5
+ * superseded by a later result for the same target (same file read again,
6
+ * same search re-run) or invalidated by a later successful edit/write to a
7
+ * covered file are pruned in preference to merely-old results. Protect-window
8
+ * and minimum-savings hysteresis semantics are unchanged.
3
9
  */
4
- import type { SessionEntry } from "./entries";
10
+ import type { SessionEntry, SessionMessageEntry } from "./entries";
5
11
  export interface PruneConfig {
6
12
  /** Keep the most recent tool output tokens intact. */
7
13
  protectTokens: number;
@@ -9,10 +15,23 @@ export interface PruneConfig {
9
15
  minimumSavings: number;
10
16
  /** Tool names that should never be pruned. */
11
17
  protectedTools: string[];
18
+ /**
19
+ * Tools in `protectedTools` whose protection is waived once the result is
20
+ * superseded (a later result for the same target, or a later successful
21
+ * edit/write to the covered file). The most recent result per target is
22
+ * never considered superseded. Optional; defaults to none.
23
+ */
24
+ staleOverridableTools?: string[];
12
25
  }
13
26
  export declare const DEFAULT_PRUNE_CONFIG: PruneConfig;
14
27
  export interface PruneResult {
15
28
  prunedCount: number;
16
29
  tokensSaved: number;
30
+ /**
31
+ * The mutated message entries. Callers whose entry source returns
32
+ * materialized copies (not live references) must write these back into
33
+ * their canonical store by id.
34
+ */
35
+ prunedEntries: SessionMessageEntry[];
17
36
  }
18
37
  export declare function pruneToolOutputs(entries: SessionEntry[], config?: PruneConfig): PruneResult;
@@ -154,6 +154,10 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
154
154
  * Callers may abort synchronously to stop consuming buffered provider events.
155
155
  */
156
156
  onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
157
+ /** Called for non-content tool-choice incapability stream events. */
158
+ onToolChoiceIncapability?: (event: Extract<AssistantMessageEvent, {
159
+ type: "toolChoiceIncapability";
160
+ }>) => void;
157
161
  /**
158
162
  * Called when GPT-5 Harmony protocol leakage is detected and mitigated.
159
163
  */
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/agent-core",
4
- "version": "0.4.4",
4
+ "version": "0.5.0",
5
5
  "description": "General-purpose agent with transport abstraction, state management, and attachment support",
6
6
  "homepage": "https://gaebal-gajae.dev",
7
7
  "author": "Yeachan-Heo",
@@ -35,9 +35,9 @@
35
35
  "fmt": "biome format --write ."
36
36
  },
37
37
  "dependencies": {
38
- "@gajae-code/ai": "0.4.4",
39
- "@gajae-code/natives": "0.4.4",
40
- "@gajae-code/utils": "0.4.4",
38
+ "@gajae-code/ai": "0.5.0",
39
+ "@gajae-code/natives": "0.5.0",
40
+ "@gajae-code/utils": "0.5.0",
41
41
  "@opentelemetry/api": "^1.9.0"
42
42
  },
43
43
  "devDependencies": {
package/src/agent-loop.ts CHANGED
@@ -815,6 +815,10 @@ async function streamAssistantResponse(
815
815
  stream.push({ type: "message_start", message: { ...partialMessage } });
816
816
  break;
817
817
 
818
+ case "toolChoiceIncapability":
819
+ config.onToolChoiceIncapability?.(event);
820
+ break;
821
+
818
822
  case "text_start":
819
823
  case "text_delta":
820
824
  case "text_end":
package/src/agent.ts CHANGED
@@ -151,6 +151,8 @@ export interface AgentOptions {
151
151
  * Use this when abort decisions must happen before buffered events continue flowing.
152
152
  */
153
153
  onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
154
+ /** Called for non-content tool-choice incapability stream events. */
155
+ onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
154
156
 
155
157
  /**
156
158
  * Called when GPT-5 Harmony protocol leakage is detected and mitigated.
@@ -309,6 +311,7 @@ export class Agent {
309
311
  #onResponse?: SimpleStreamOptions["onResponse"];
310
312
  #onSseEvent?: SimpleStreamOptions["onSseEvent"];
311
313
  #onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
314
+ #onToolChoiceIncapability?: AgentLoopConfig["onToolChoiceIncapability"];
312
315
  #onHarmonyLeak?: (event: HarmonyAuditEvent) => void | Promise<void>;
313
316
  #onBeforeYield?: () => Promise<void> | void;
314
317
  #shouldPause?: AgentLoopConfig["shouldPause"];
@@ -373,6 +376,7 @@ export class Agent {
373
376
  this.#intentTracing = opts.intentTracing === true;
374
377
  this.#getToolChoice = opts.getToolChoice;
375
378
  this.#onAssistantMessageEvent = opts.onAssistantMessageEvent;
379
+ this.#onToolChoiceIncapability = opts.onToolChoiceIncapability;
376
380
  this.#onHarmonyLeak = opts.onHarmonyLeak;
377
381
  this.#shouldPause = opts.shouldPause;
378
382
  this.beforeToolCall = opts.beforeToolCall;
@@ -680,7 +684,11 @@ export class Agent {
680
684
  if (!source) return undefined;
681
685
 
682
686
  const guarded: CursorExecHandlers = {};
683
- const read = source.read;
687
+ // Bind each handler to `source`: they are methods of a CursorExecHandlers
688
+ // instance that reference private fields via `this`. Extracting them bare
689
+ // (`const read = source.read`) and calling `read(args)` would invoke them with
690
+ // `this === undefined`, throwing "undefined is not an object (this.#optionsForCall)".
691
+ const read = source.read?.bind(source);
684
692
  if (read) {
685
693
  guarded.read = async args => {
686
694
  this.#assertActiveRun(runId);
@@ -689,7 +697,7 @@ export class Agent {
689
697
  return result;
690
698
  };
691
699
  }
692
- const ls = source.ls;
700
+ const ls = source.ls?.bind(source);
693
701
  if (ls) {
694
702
  guarded.ls = async args => {
695
703
  this.#assertActiveRun(runId);
@@ -698,7 +706,7 @@ export class Agent {
698
706
  return result;
699
707
  };
700
708
  }
701
- const grep = source.grep;
709
+ const grep = source.grep?.bind(source);
702
710
  if (grep) {
703
711
  guarded.grep = async args => {
704
712
  this.#assertActiveRun(runId);
@@ -707,7 +715,7 @@ export class Agent {
707
715
  return result;
708
716
  };
709
717
  }
710
- const write = source.write;
718
+ const write = source.write?.bind(source);
711
719
  if (write) {
712
720
  guarded.write = async args => {
713
721
  this.#assertActiveRun(runId);
@@ -716,7 +724,7 @@ export class Agent {
716
724
  return result;
717
725
  };
718
726
  }
719
- const deleteHandler = source.delete;
727
+ const deleteHandler = source.delete?.bind(source);
720
728
  if (deleteHandler) {
721
729
  guarded.delete = async args => {
722
730
  this.#assertActiveRun(runId);
@@ -725,7 +733,7 @@ export class Agent {
725
733
  return result;
726
734
  };
727
735
  }
728
- const shell = source.shell;
736
+ const shell = source.shell?.bind(source);
729
737
  if (shell) {
730
738
  guarded.shell = async args => {
731
739
  this.#assertActiveRun(runId);
@@ -734,7 +742,7 @@ export class Agent {
734
742
  return result;
735
743
  };
736
744
  }
737
- const shellStream = source.shellStream;
745
+ const shellStream = source.shellStream?.bind(source);
738
746
  if (shellStream) {
739
747
  guarded.shellStream = async (args, callbacks) => {
740
748
  this.#assertActiveRun(runId);
@@ -743,7 +751,7 @@ export class Agent {
743
751
  return result;
744
752
  };
745
753
  }
746
- const diagnostics = source.diagnostics;
754
+ const diagnostics = source.diagnostics?.bind(source);
747
755
  if (diagnostics) {
748
756
  guarded.diagnostics = async args => {
749
757
  this.#assertActiveRun(runId);
@@ -752,7 +760,7 @@ export class Agent {
752
760
  return result;
753
761
  };
754
762
  }
755
- const mcp = source.mcp;
763
+ const mcp = source.mcp?.bind(source);
756
764
  if (mcp) {
757
765
  guarded.mcp = async call => {
758
766
  this.#assertActiveRun(runId);
@@ -1218,6 +1226,12 @@ export class Agent {
1218
1226
  this.#onAssistantMessageEvent?.(message, event);
1219
1227
  }
1220
1228
  : undefined,
1229
+ onToolChoiceIncapability: this.#onToolChoiceIncapability
1230
+ ? event => {
1231
+ if (this.#activeRunId !== runId) return;
1232
+ this.#onToolChoiceIncapability?.(event);
1233
+ }
1234
+ : undefined,
1221
1235
  onHarmonyLeak: this.#onHarmonyLeak,
1222
1236
  getToolChoice,
1223
1237
  getReasoning: () => this.#state.thinkingLevel,
@@ -46,6 +46,9 @@ export interface BuildOptions {
46
46
  export class StablePrefix {
47
47
  #snapshot: StablePrefixSnapshot | null = null;
48
48
  #version = 0;
49
+ #sourceSystemPrompt: readonly string[] | null = null;
50
+ #sourceTools: AgentContext["tools"] | null = null;
51
+ #sourceIntentTracing: boolean | null = null;
49
52
 
50
53
  get fingerprint(): string {
51
54
  return this.#snapshot?.fingerprint ?? "<unbuilt>";
@@ -65,6 +68,9 @@ export class StablePrefix {
65
68
  const systemPrompt = cloneJson(snapshot.systemPrompt);
66
69
  const tools = normalizeImportedTools(snapshot.tools, options);
67
70
  const fingerprint = computeFingerprint(systemPrompt, tools, options);
71
+ this.#sourceSystemPrompt = null;
72
+ this.#sourceTools = null;
73
+ this.#sourceIntentTracing = null;
68
74
  if (fingerprint !== snapshot.fingerprint) {
69
75
  throw new Error(
70
76
  `StablePrefix.importSnapshot() fingerprint mismatch: expected ${fingerprint}, received ${snapshot.fingerprint}`,
@@ -79,11 +85,26 @@ export class StablePrefix {
79
85
  * Returns `true` if the prefix actually changed (cache miss imminent).
80
86
  */
81
87
  build(context: AgentContext, options: BuildOptions): boolean {
88
+ if (
89
+ this.#snapshot &&
90
+ this.#sourceSystemPrompt === context.systemPrompt &&
91
+ this.#sourceTools === context.tools &&
92
+ this.#sourceIntentTracing === options.intentTracing
93
+ ) {
94
+ const sourceFingerprint = takeSnapshot(context, options).fingerprint;
95
+ if (this.#snapshot.fingerprint === sourceFingerprint) return false;
96
+ }
82
97
  const snapshot = takeSnapshot(context, options);
83
98
  if (this.#snapshot && this.#snapshot.fingerprint === snapshot.fingerprint) {
99
+ this.#sourceSystemPrompt = context.systemPrompt;
100
+ this.#sourceTools = context.tools;
101
+ this.#sourceIntentTracing = options.intentTracing;
84
102
  return false;
85
103
  }
86
104
  this.#snapshot = snapshot;
105
+ this.#sourceSystemPrompt = context.systemPrompt;
106
+ this.#sourceTools = context.tools;
107
+ this.#sourceIntentTracing = options.intentTracing;
87
108
  this.#version++;
88
109
  return true;
89
110
  }
@@ -91,6 +112,9 @@ export class StablePrefix {
91
112
  /** Force rebuild on the next `build()` call. */
92
113
  invalidate(): void {
93
114
  this.#snapshot = null;
115
+ this.#sourceSystemPrompt = null;
116
+ this.#sourceTools = null;
117
+ this.#sourceIntentTracing = null;
94
118
  }
95
119
 
96
120
  /**
@@ -175,8 +199,8 @@ export class AppendOnlyContextManager {
175
199
  readonly log = new AppendOnlyLog();
176
200
  /** How many normalized messages were synced into the log as of the last sync. */
177
201
  #lastSyncCount = 0;
178
- /** Rolling digest of synced message content — detects in-place rewrites. */
179
- #syncedDigest = 0;
202
+ /** Fingerprint plus source bytes of synced message content — detects in-place rewrites with no hash-only equality. */
203
+ #syncedDigest = emptyMessageDigest();
180
204
  /** Number of provider-normalized messages that were seeded before child-local messages. */
181
205
  #seededPrefixCount = 0;
182
206
 
@@ -208,19 +232,22 @@ export class AppendOnlyContextManager {
208
232
  * (same length, changed content via a rolling digest).
209
233
  */
210
234
  syncMessages(normalizedMessages: any[]): void {
211
- const seededPrefix = this.#seededPrefixCount > 0 ? this.log.toMessages().slice(0, this.#seededPrefixCount) : [];
235
+ const seededPrefixLength = this.#seededPrefixCount;
212
236
  const includesSeedPrefix =
213
- seededPrefix.length > 0 &&
214
- normalizedMessages.length >= seededPrefix.length &&
215
- this.#computeDigest(normalizedMessages.slice(0, seededPrefix.length)) === this.#computeDigest(seededPrefix);
237
+ seededPrefixLength > 0 &&
238
+ normalizedMessages.length >= seededPrefixLength &&
239
+ this.#computeDigestRange(normalizedMessages, 0, seededPrefixLength).source ===
240
+ this.#computeDigestRange(this.log.entries(), 0, seededPrefixLength).source;
216
241
  const messagesToSync =
217
- seededPrefix.length > 0 && !includesSeedPrefix ? [...seededPrefix, ...normalizedMessages] : normalizedMessages;
242
+ seededPrefixLength > 0 && !includesSeedPrefix
243
+ ? [...this.log.entries().slice(0, seededPrefixLength), ...normalizedMessages]
244
+ : normalizedMessages;
218
245
 
219
246
  // Detect in-place rewrites of already-synced messages.
220
247
  if (
221
248
  this.#lastSyncCount > 0 &&
222
249
  this.#lastSyncCount <= messagesToSync.length &&
223
- this.#computeDigest(messagesToSync.slice(0, this.#lastSyncCount)) !== this.#syncedDigest
250
+ this.#computeDigestRange(messagesToSync, 0, this.#lastSyncCount).source !== this.#syncedDigest.source
224
251
  ) {
225
252
  if (this.#seededPrefixCount > 0) {
226
253
  throw new Error("AppendOnlyContextManager.syncMessages() seed prefix changed");
@@ -266,7 +293,7 @@ export class AppendOnlyContextManager {
266
293
  this.prefix.invalidate();
267
294
  this.log.clear();
268
295
  this.#lastSyncCount = 0;
269
- this.#syncedDigest = 0;
296
+ this.#syncedDigest = emptyMessageDigest();
270
297
  this.#seededPrefixCount = 0;
271
298
  }
272
299
 
@@ -274,7 +301,7 @@ export class AppendOnlyContextManager {
274
301
  resetSyncCursor(): void {
275
302
  this.log.clear();
276
303
  this.#lastSyncCount = 0;
277
- this.#syncedDigest = 0;
304
+ this.#syncedDigest = emptyMessageDigest();
278
305
  this.#seededPrefixCount = 0;
279
306
  }
280
307
 
@@ -294,36 +321,28 @@ export class AppendOnlyContextManager {
294
321
  this.prefix.invalidate();
295
322
  this.log.clear();
296
323
  this.#lastSyncCount = 0;
297
- this.#syncedDigest = 0;
324
+ this.#syncedDigest = emptyMessageDigest();
298
325
  this.#seededPrefixCount = 0;
299
326
  this.prefix.build(context, options);
300
327
  }
301
328
 
302
329
  /**
303
- * Deterministic digest over every field the provider may serialize — role,
304
- * content, tool calls (both `toolCalls` and OpenAI-wire `tool_calls`),
305
- * `tool_call_id`, `name`, `id`. Hashed with the same FNV-style rolling
306
- * accumulator so in-place rewrites of *any* of these fields are visible.
330
+ * Deterministic digest over the provider-visible message payload. The source
331
+ * string is kept and compared for equality so the hash is only a fast summary,
332
+ * never the authority for accepting append-only sync state.
307
333
  */
308
- #computeDigest(messages: readonly unknown[]): number {
309
- let hash = 0;
310
- for (let i = 0; i < messages.length; i++) {
311
- const msg = messages[i];
312
- if (!msg || typeof msg !== "object") continue;
313
- const m = msg as Record<string, unknown>;
314
- const payload = JSON.stringify({
315
- r: m.role ?? null,
316
- c: m.content ?? null,
317
- tc: m.toolCalls ?? m.tool_calls ?? null,
318
- tcid: m.tool_call_id ?? null,
319
- n: m.name ?? null,
320
- id: m.id ?? null,
321
- });
322
- for (let j = 0; j < payload.length; j++) {
323
- hash = ((hash << 5) - hash + payload.charCodeAt(j)) | 0;
324
- }
334
+ #computeDigest(messages: readonly unknown[]): MessageDigest {
335
+ return this.#computeDigestRange(messages, 0, messages.length);
336
+ }
337
+
338
+ #computeDigestRange(messages: readonly unknown[], start: number, end: number): MessageDigest {
339
+ let source = "[";
340
+ for (let i = start; i < end; i++) {
341
+ if (i > start) source += ",";
342
+ source += JSON.stringify(messages[i]) ?? "null";
325
343
  }
326
- return hash >>> 0;
344
+ source += "]";
345
+ return { hash: hashSource(source), source };
327
346
  }
328
347
  }
329
348
 
@@ -331,6 +350,27 @@ export class AppendOnlyContextManager {
331
350
  // Snapshot helpers
332
351
  // ---------------------------------------------------------------------------
333
352
 
353
+ type MessageDigest = {
354
+ hash: number | bigint;
355
+ source: string;
356
+ };
357
+
358
+ function emptyMessageDigest(): MessageDigest {
359
+ return { hash: hashSource("[]"), source: "[]" };
360
+ }
361
+
362
+ function hashSource(source: string): number | bigint {
363
+ return typeof Bun !== "undefined" ? Bun.hash(source) : hashString32(source);
364
+ }
365
+
366
+ function hashString32(value: string): number {
367
+ let hash = 0;
368
+ for (let i = 0; i < value.length; i++) {
369
+ hash = ((hash << 5) - hash + value.charCodeAt(i)) | 0;
370
+ }
371
+ return hash >>> 0;
372
+ }
373
+
334
374
  function takeSnapshot(context: AgentContext, options: BuildOptions): StablePrefixSnapshot {
335
375
  const systemPrompt = [...context.systemPrompt];
336
376
  const tools = normalizeTools(context.tools, options.intentTracing) ?? [];
@@ -347,8 +387,42 @@ function normalizeImportedTools(tools: readonly Tool[], options: BuildOptions):
347
387
  return cloneJson(normalizedTools);
348
388
  }
349
389
 
350
- function cloneJson<T>(value: T): T {
351
- return JSON.parse(JSON.stringify(value)) as T;
390
+ export function cloneJson<T>(value: T): T {
391
+ return cloneJsonValue(value) as T;
392
+ }
393
+
394
+ function cloneJsonValue(value: unknown, key = "", applyToJson = true): unknown {
395
+ if (value === null) return null;
396
+ const type = typeof value;
397
+ if (type === "number") return Number.isFinite(value) ? value : null;
398
+ // JSON.stringify drops function/symbol/undefined values (object props
399
+ // omitted, array elements become null via the array walk below).
400
+ if (type === "undefined" || type === "function" || type === "symbol") return undefined;
401
+ if (type !== "object") return value;
402
+ if (applyToJson) {
403
+ // JSON.stringify performs a single Get of `toJSON` per holder/key and
404
+ // serializes the returned replacement WITHOUT re-dispatching the
405
+ // replacement's own toJSON at the same level (nested properties still
406
+ // dispatch normally). Mirror that exactly to keep byte parity.
407
+ const toJSON = (value as { toJSON?: unknown }).toJSON;
408
+ if (typeof toJSON === "function") {
409
+ return cloneJsonValue(toJSON.call(value, key), key, false);
410
+ }
411
+ }
412
+ if (Array.isArray(value)) {
413
+ const cloned: unknown[] = new Array(value.length);
414
+ for (let i = 0; i < value.length; i++) {
415
+ const item = Object.hasOwn(value, i) ? cloneJsonValue(value[i], String(i)) : undefined;
416
+ cloned[i] = item === undefined ? null : item;
417
+ }
418
+ return cloned;
419
+ }
420
+ const cloned: Record<string, unknown> = {};
421
+ for (const key of Object.keys(value as object)) {
422
+ const clonedValue = cloneJsonValue((value as Record<string, unknown>)[key], key);
423
+ if (clonedValue !== undefined) cloned[key] = clonedValue;
424
+ }
425
+ return cloned;
352
426
  }
353
427
 
354
428
  function computeFingerprint(systemPrompt: string[], tools: Tool[], options: BuildOptions): string {
@@ -13,7 +13,6 @@ import {
13
13
  type Model,
14
14
  type Usage,
15
15
  } from "@gajae-code/ai";
16
- import { countTokens } from "@gajae-code/natives";
17
16
  import { logger, prompt } from "@gajae-code/utils";
18
17
  import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
19
18
  import type { AgentMessage, AgentTool } from "../types";
@@ -268,18 +267,97 @@ export function resolveThresholdTokens(
268
267
  const IMAGE_TOKEN_ESTIMATE = 1200;
269
268
 
270
269
  /**
271
- * Estimate token count for a message using cl100k_base via the native
272
- * tokenizer. This is not Anthropic's first-party tokenizer (Anthropic doesn't
273
- * publish one) but is within ~5–10% across English/code text.
270
+ * Lazily-required native `countTokens`. `@gajae-code/natives` dlopens a ~39MB
271
+ * addon; importing it at module scope would put that cost on every cold path
272
+ * that touches compaction exports (status line, print mode, context report).
273
+ * Deferring the require to the first context-changing call keeps the trivial
274
+ * `-p` / display paths native-free.
274
275
  */
275
- export function estimateTokens(message: AgentMessage): number {
276
+ let cachedNativeCountTokens: ((input: string | string[], encoding?: unknown) => number) | null = null;
277
+
278
+ function nativeCountTokens(fragments: string[]): number {
279
+ if (!cachedNativeCountTokens) {
280
+ const { createRequire } = require("node:module") as typeof import("node:module");
281
+ const requireFromHere = createRequire(import.meta.url);
282
+ const natives = requireFromHere("@gajae-code/natives") as {
283
+ countTokens: (input: string | string[], encoding?: unknown) => number;
284
+ };
285
+ cachedNativeCountTokens = natives.countTokens;
286
+ }
287
+ return cachedNativeCountTokens(fragments);
288
+ }
289
+
290
+ function countCollectedMessageFragments(collected: { fragments: string[]; extra: number }): number {
291
+ return nativeCountTokens(collected.fragments) + collected.extra;
292
+ }
293
+
294
+ /**
295
+ * Estimate token count for a message using the native o200k tokenizer.
296
+ * Exact for o200k only; an approximation for Anthropic/other model families
297
+ * (Anthropic doesn't publish a tokenizer) within ~5–10% on English/code text.
298
+ *
299
+ * This materializes the native BPE table (~50MB RSS) on first call. Use it
300
+ * only for context-changing decisions (compaction trigger/cut points, pruning
301
+ * budgets, branch summarization, fork-context seeding, context-limit
302
+ * enforcement). For display-only totals use
303
+ * {@link estimateMessageTokensHeuristic}.
304
+ */
305
+ export function countMessageTokensNativeO200k(message: AgentMessage): number {
306
+ return countCollectedMessageFragments(collectMessageFragments(message));
307
+ }
308
+
309
+ /**
310
+ * Backwards-compatible alias for {@link countMessageTokensNativeO200k}.
311
+ * Existing callers treat this as the canonical message-token estimator for
312
+ * context-changing decisions.
313
+ */
314
+ export const estimateTokens = countMessageTokensNativeO200k;
315
+
316
+ /**
317
+ * Average bytes per token for the cheap heuristic. ~4 bytes/token is the
318
+ * conventional approximation for English/code text under modern BPE
319
+ * vocabularies; it intentionally errs slightly low-precision in exchange for
320
+ * never touching the native tokenizer (and its ~50MB BPE table).
321
+ */
322
+ const HEURISTIC_BYTES_PER_TOKEN = 4;
323
+
324
+ /**
325
+ * Cheap, native-free token estimate for a message. Suitable ONLY for
326
+ * display/init surfaces (status line, /context report, HUD totals) — never
327
+ * for context-changing decisions, which must use
328
+ * {@link countMessageTokensNativeO200k}.
329
+ */
330
+ export function estimateMessageTokensHeuristic(message: AgentMessage): number {
331
+ const { fragments, extra } = collectMessageFragments(message);
332
+ let bytes = 0;
333
+ for (const fragment of fragments) {
334
+ bytes += fragment.length;
335
+ }
336
+ return extra + Math.ceil(bytes / HEURISTIC_BYTES_PER_TOKEN);
337
+ }
338
+
339
+ /**
340
+ * Cheap, native-free token estimate for plain string fragments. Display-only
341
+ * counterpart of the native `countTokens(fragments)` aggregate.
342
+ */
343
+ export function estimateTextTokensHeuristic(fragments: string | readonly string[]): number {
344
+ if (typeof fragments === "string") return Math.ceil(fragments.length / HEURISTIC_BYTES_PER_TOKEN);
345
+ let bytes = 0;
346
+ for (const fragment of fragments) {
347
+ bytes += fragment.length;
348
+ }
349
+ return Math.ceil(bytes / HEURISTIC_BYTES_PER_TOKEN);
350
+ }
351
+
352
+ /** Shared content walk for both the native and heuristic estimators. */
353
+ function collectMessageFragments(message: AgentMessage): { fragments: string[]; extra: number } {
276
354
  const fragments: string[] = [];
277
355
  let extra = 0;
278
356
  if ((message as { role?: string }).role === "bashExecution") {
279
357
  const bash = message as { command?: unknown; output?: unknown };
280
358
  if (typeof bash.command === "string") fragments.push(bash.command);
281
359
  if (typeof bash.output === "string") fragments.push(bash.output);
282
- return fragments.length === 0 ? 0 : countTokens(fragments);
360
+ return { fragments, extra };
283
361
  }
284
362
 
285
363
  switch (message.role) {
@@ -331,20 +409,45 @@ export function estimateTokens(message: AgentMessage): number {
331
409
  break;
332
410
  }
333
411
  default:
334
- return 0;
412
+ break;
413
+ }
414
+
415
+ return { fragments, extra };
416
+ }
417
+
418
+ function entryTokenFingerprint(
419
+ entry: SessionEntry,
420
+ message: AgentMessage,
421
+ collected: { fragments: string[]; extra: number },
422
+ ): string {
423
+ const maybePruned = message as { prunedAt?: unknown };
424
+ let fingerprint = `${entry.type.length}:${entry.type}${(entry.id ?? "").length}:${entry.id ?? ""}${message.role.length}:${message.role}${String(collected.extra).length}:${String(collected.extra)}${collected.fragments.length}:`;
425
+ for (const fragment of collected.fragments) fingerprint += `${fragment.length}:${fragment}`;
426
+ if (maybePruned.prunedAt !== undefined) {
427
+ const prunedAt = String(maybePruned.prunedAt);
428
+ fingerprint += `prunedAt${prunedAt.length}:${prunedAt}`;
335
429
  }
430
+ return fingerprint;
431
+ }
336
432
 
337
- if (fragments.length === 0) return extra;
338
- return extra + countTokens(fragments);
433
+ const entryTokenCache = new WeakMap<SessionEntry, { fingerprint: string; tokens: number }>();
434
+
435
+ export function estimateEntryTokens(entry: SessionEntry): number {
436
+ const msg = getMessageFromEntry(entry);
437
+ if (!msg) return 0;
438
+ const collected = collectMessageFragments(msg);
439
+ const fingerprint = entryTokenFingerprint(entry, msg, collected);
440
+ const cached = entryTokenCache.get(entry);
441
+ if (cached?.fingerprint === fingerprint) return cached.tokens;
442
+ const tokens = countCollectedMessageFragments(collected);
443
+ entryTokenCache.set(entry, { fingerprint, tokens });
444
+ return tokens;
339
445
  }
340
446
 
341
- function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
447
+ export function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
342
448
  let total = 0;
343
449
  for (let i = startIndex; i < endIndex; i++) {
344
- const msg = getMessageFromEntry(entries[i]);
345
- if (msg) {
346
- total += estimateTokens(msg);
347
- }
450
+ total += estimateEntryTokens(entries[i]);
348
451
  }
349
452
  return total;
350
453
  }
@@ -461,18 +564,23 @@ export function findCutPoint(
461
564
  if (entry.type !== "message") continue;
462
565
 
463
566
  // Estimate this message's size
464
- const messageTokens = estimateTokens(entry.message);
567
+ const messageTokens = estimateEntryTokens(entry);
465
568
  accumulatedTokens += messageTokens;
466
569
 
467
570
  // Check if we've exceeded the budget
468
571
  if (accumulatedTokens >= keepRecentTokens) {
469
572
  // Find the closest valid cut point at or after this entry
573
+ let foundCutPoint = false;
470
574
  for (let c = 0; c < cutPoints.length; c++) {
471
575
  if (cutPoints[c] >= i) {
472
576
  cutIndex = cutPoints[c];
577
+ foundCutPoint = true;
473
578
  break;
474
579
  }
475
580
  }
581
+ if (!foundCutPoint) {
582
+ cutIndex = cutPoints[cutPoints.length - 1];
583
+ }
476
584
  break;
477
585
  }
478
586
  }
@@ -24,6 +24,12 @@ export interface ModelChangeEntry extends SessionEntryBase {
24
24
  model: string;
25
25
  /** Role: "default", "smol", "slow", etc. Undefined treated as "default" */
26
26
  role?: string;
27
+ /** Requested model before a runtime substitution/fallback, in "provider/modelId" format. */
28
+ previousModel?: string;
29
+ /** Machine-readable reason for runtime model substitution/fallback. */
30
+ reason?: string;
31
+ /** Effective thinking level when the change was recorded. */
32
+ thinkingLevel?: string | null;
27
33
  }
28
34
 
29
35
  export interface ServiceTierChangeEntry extends SessionEntryBase {
@@ -154,7 +154,7 @@ export function withOpenAiRemoteCompactionPreserveData(
154
154
  // Input/output filtering for OpenAI compact endpoint
155
155
  // ============================================================================
156
156
 
157
- function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number {
157
+ export function estimateOpenAiCompactInputTokens(input: Array<Record<string, unknown>>, instructions: string): number {
158
158
  let chars = instructions.length;
159
159
  for (const item of input) {
160
160
  chars += JSON.stringify(item).length;
@@ -200,22 +200,31 @@ function shouldKeepOpenAiCompactOutputItem(item: Record<string, unknown>): boole
200
200
  return shouldKeepOpenAiCompactOutputUserMessage(item);
201
201
  }
202
202
 
203
- function trimOpenAiCompactInput(
203
+ export function trimOpenAiCompactInput(
204
204
  input: Array<Record<string, unknown>>,
205
205
  contextWindow: number,
206
206
  instructions: string,
207
207
  ): Array<Record<string, unknown>> {
208
+ const itemLengths = input.map(item => JSON.stringify(item).length);
209
+ let chars = instructions.length;
210
+ for (const length of itemLengths) chars += length;
211
+
212
+ function removeAt(index: number): void {
213
+ chars -= itemLengths[index] ?? 0;
214
+ trimmed.splice(index, 1);
215
+ itemLengths.splice(index, 1);
216
+ }
208
217
  const trimmed = [...input];
209
- while (trimmed.length > 0 && estimateOpenAiCompactInputTokens(trimmed, instructions) > contextWindow) {
218
+ while (trimmed.length > 0 && Math.ceil(chars / 4) > contextWindow) {
210
219
  const last = trimmed[trimmed.length - 1];
211
220
  if (last?.type === "function_call_output" || last?.type === "custom_tool_call_output") {
212
221
  const callId = typeof last.call_id === "string" ? last.call_id : undefined;
213
222
  const callType = last.type === "custom_tool_call_output" ? "custom_tool_call" : "function_call";
214
- trimmed.pop();
223
+ removeAt(trimmed.length - 1);
215
224
  if (callId) {
216
225
  const matchingCallIndex = trimmed.findLastIndex(item => item.type === callType && item.call_id === callId);
217
226
  if (matchingCallIndex >= 0) {
218
- trimmed.splice(matchingCallIndex, 1);
227
+ removeAt(matchingCallIndex);
219
228
  }
220
229
  }
221
230
  continue;
@@ -223,7 +232,7 @@ function trimOpenAiCompactInput(
223
232
  if (!last || !shouldTrimOpenAiCompactInputItem(last)) {
224
233
  break;
225
234
  }
226
- trimmed.pop();
235
+ removeAt(trimmed.length - 1);
227
236
  }
228
237
  return trimmed;
229
238
  }
@@ -1,10 +1,16 @@
1
1
  /**
2
2
  * Tool output pruning utilities for compaction.
3
+ *
4
+ * Candidate selection is staleness-aware: tool results that have been
5
+ * superseded by a later result for the same target (same file read again,
6
+ * same search re-run) or invalidated by a later successful edit/write to a
7
+ * covered file are pruned in preference to merely-old results. Protect-window
8
+ * and minimum-savings hysteresis semantics are unchanged.
3
9
  */
4
10
 
5
- import type { ToolResultMessage } from "@gajae-code/ai";
11
+ import type { ToolCall, ToolResultMessage } from "@gajae-code/ai";
6
12
  import type { AgentMessage } from "../types";
7
- import { estimateTokens } from "./compaction";
13
+ import { estimateEntryTokens } from "./compaction";
8
14
  import type { SessionEntry, SessionMessageEntry } from "./entries";
9
15
 
10
16
  export interface PruneConfig {
@@ -14,23 +20,100 @@ export interface PruneConfig {
14
20
  minimumSavings: number;
15
21
  /** Tool names that should never be pruned. */
16
22
  protectedTools: string[];
23
+ /**
24
+ * Tools in `protectedTools` whose protection is waived once the result is
25
+ * superseded (a later result for the same target, or a later successful
26
+ * edit/write to the covered file). The most recent result per target is
27
+ * never considered superseded. Optional; defaults to none.
28
+ */
29
+ staleOverridableTools?: string[];
17
30
  }
18
31
 
19
32
  export const DEFAULT_PRUNE_CONFIG: PruneConfig = {
20
33
  protectTokens: 40_000,
21
34
  minimumSavings: 20_000,
22
35
  protectedTools: ["skill", "read"],
36
+ staleOverridableTools: ["read"],
23
37
  };
24
38
 
25
39
  export interface PruneResult {
26
40
  prunedCount: number;
27
41
  tokensSaved: number;
42
+ /**
43
+ * The mutated message entries. Callers whose entry source returns
44
+ * materialized copies (not live references) must write these back into
45
+ * their canonical store by id.
46
+ */
47
+ prunedEntries: SessionMessageEntry[];
28
48
  }
29
49
 
30
- function createPrunedNotice(tokens: number): string {
50
+ const DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER = 1.25;
51
+
52
+ function createGenericPrunedNotice(tokens: number): string {
31
53
  return `[Output truncated - ${tokens} tokens]`;
32
54
  }
33
55
 
56
+ function firstTextContent(message: ToolResultMessage): string {
57
+ if (typeof message.content === "string") return message.content;
58
+ const block = message.content.find(part => part.type === "text");
59
+ return block?.type === "text" ? block.text : "";
60
+ }
61
+
62
+ function firstErrorLine(text: string): string | undefined {
63
+ return text
64
+ .split(/\r?\n/)
65
+ .find(line => /error|failed|exception|panic/i.test(line))
66
+ ?.trim();
67
+ }
68
+
69
+ function truncateField(value: string, maxLength: number): string {
70
+ if (value.length <= maxLength) return value;
71
+ if (maxLength <= 1) return "…";
72
+ return `${value.slice(0, maxLength - 1)}…`;
73
+ }
74
+
75
+ function resultDigest(message: ToolResultMessage): string | undefined {
76
+ const toolName = message.toolName.toLowerCase();
77
+ const text = firstTextContent(message);
78
+ if (toolName === "bash") {
79
+ const details = message as { details?: { exitCode?: unknown } };
80
+ const exitCode =
81
+ typeof details.details?.exitCode === "number" ? details.details.exitCode : message.isError ? 1 : 0;
82
+ const tail = text.trim().split(/\r?\n/).filter(Boolean).at(-1) ?? "";
83
+ const error = firstErrorLine(text);
84
+ return [`exit=${exitCode}`, tail ? `tail=${tail}` : undefined, error ? `error=${error}` : undefined]
85
+ .filter((part): part is string => part !== undefined)
86
+ .join("; ");
87
+ }
88
+ if (toolName === "search" || toolName === "grep") {
89
+ const match = text.match(/(\d+)\s+matches?/i) ?? text.match(/totalMatches["']?:\s*(\d+)/i);
90
+ const files = text.match(/(\d+)\s+files?/i) ?? text.match(/filesWithMatches["']?:\s*(\d+)/i);
91
+ const error = firstErrorLine(text);
92
+ return (
93
+ [
94
+ match ? `matches=${match[1]}` : undefined,
95
+ files ? `files=${files[1]}` : undefined,
96
+ error ? `error=${error}` : undefined,
97
+ ]
98
+ .filter((part): part is string => part !== undefined)
99
+ .join("; ") || "search digest unavailable"
100
+ );
101
+ }
102
+ return undefined;
103
+ }
104
+
105
+ function createPrunedNotice(tokens: number, message?: ToolResultMessage): string {
106
+ const generic = createGenericPrunedNotice(tokens);
107
+ const digest = message ? resultDigest(message) : undefined;
108
+ if (!digest) return generic;
109
+ const genericTokens = Math.ceil(generic.length / 4);
110
+ const maxTokens = Math.max(genericTokens, Math.floor(genericTokens * DIGEST_NOTICE_TOKEN_CAP_MULTIPLIER));
111
+ const prefix = `[Output truncated - ${tokens} tokens; `;
112
+ const suffix = "]";
113
+ const maxChars = Math.max(0, maxTokens * 4 - prefix.length - suffix.length);
114
+ return `${prefix}${truncateField(digest, maxChars)}${suffix}`;
115
+ }
116
+
34
117
  function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefined {
35
118
  if (entry.type !== "message") return undefined;
36
119
  const message = entry.message as AgentMessage;
@@ -38,55 +121,311 @@ function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefine
38
121
  return message as ToolResultMessage;
39
122
  }
40
123
 
41
- function estimatePrunedSavings(tokens: number): number {
42
- const noticeTokens = Math.ceil(createPrunedNotice(tokens).length / 4);
124
+ function estimatePrunedSavings(tokens: number, notice: string): number {
125
+ const noticeTokens = Math.ceil(notice.length / 4);
43
126
  return Math.max(0, tokens - noticeTokens);
44
127
  }
45
128
 
129
+ const EDIT_TOOL_NAMES = new Set(["edit", "write", "apply_patch", "ast_edit"]);
130
+
131
+ /** Extract the file-path argument from a tool call, when the tool has one. */
132
+ function toolCallPath(call: ToolCall): string | undefined {
133
+ const args = call.arguments;
134
+ const path = args.path ?? args.file_path ?? args.filePath;
135
+ return typeof path === "string" && path.length > 0 ? path : undefined;
136
+ }
137
+
138
+ /**
139
+ * `*** Add|Update|Delete File: <path>` headers open a hunk; `*** Move to:
140
+ * <path>` attaches a rename destination to the current hunk. Move
141
+ * destinations count as touched paths: a rename onto a file invalidates
142
+ * earlier reads of that destination.
143
+ */
144
+ const APPLY_PATCH_HEADER = /^\*\*\* (?:((?:Add|Update|Delete) File)|(Move to)): (.+)$/gm;
145
+
146
+ /**
147
+ * Paths touched by an edit-class tool call, grouped per hunk so a failed
148
+ * hunk can be excluded wholesale (its rename destination included). Most
149
+ * edit tools carry a single path argument; apply_patch envelopes carry an
150
+ * `input` string with per-file headers instead. The envelope shape can
151
+ * arrive under the custom `apply_patch` tool OR the regular `edit` tool
152
+ * (providers without custom-tool support fall back to the JSON function), so
153
+ * any edit-class call with a string `input` is parsed for headers.
154
+ */
155
+ function editToolPathGroups(call: ToolCall): string[][] {
156
+ const path = toolCallPath(call);
157
+ if (path !== undefined) return [[path]];
158
+ const input = call.arguments.input;
159
+ if (typeof input !== "string") return [];
160
+ const groups: string[][] = [];
161
+ for (const match of input.matchAll(APPLY_PATCH_HEADER)) {
162
+ const headerPath = match[3]?.trim();
163
+ if (!headerPath) continue;
164
+ const isMoveTo = match[2] !== undefined;
165
+ if (isMoveTo && groups.length > 0) {
166
+ groups[groups.length - 1].push(headerPath);
167
+ } else {
168
+ groups.push([headerPath]);
169
+ }
170
+ }
171
+ return groups;
172
+ }
173
+
174
+ /**
175
+ * Trailing read selectors (`:50`, `:50-200`, `:50+150`, `:5-16,960-973`,
176
+ * `:raw`, `:conflicts`), possibly stacked (`:2-4:raw`). Stripped to resolve
177
+ * the underlying file for edit invalidation.
178
+ */
179
+ const READ_SELECTOR_SUFFIX = /:(?:raw|conflicts|\d+(?:[-+]\d+)?(?:,\d+(?:[-+]\d+)?)*)$/;
180
+
181
+ /** Base file path of a read target with any line/mode selectors stripped. */
182
+ function readBasePath(path: string): string {
183
+ let base = path;
184
+ while (READ_SELECTOR_SUFFIX.test(base)) {
185
+ base = base.replace(READ_SELECTOR_SUFFIX, "");
186
+ }
187
+ return base;
188
+ }
189
+
190
+ /**
191
+ * Stable identity for "the same logical lookup": same tool re-targeting the
192
+ * same subject. A later result with the same key supersedes earlier ones.
193
+ * Keys are canonical JSON tuples so user-controlled text (patterns, paths)
194
+ * can never collide via delimiter ambiguity. Search keys include pagination
195
+ * (`skip`) and result-shaping flags (`i`, `gitignore`): a later page or a
196
+ * differently-shaped search complements earlier output, it does not replace it.
197
+ */
198
+ function toolTargetKey(call: ToolCall): string | undefined {
199
+ const path = toolCallPath(call);
200
+ if (path !== undefined) return JSON.stringify([call.name, "path", path]);
201
+ const pattern = call.arguments.pattern;
202
+ if (typeof pattern === "string" && pattern.length > 0) {
203
+ const paths = call.arguments.paths;
204
+ const pathList = Array.isArray(paths) ? paths.filter((p): p is string => typeof p === "string") : [];
205
+ const skip = typeof call.arguments.skip === "number" ? call.arguments.skip : 0;
206
+ const caseInsensitive = call.arguments.i === true;
207
+ const gitignore = call.arguments.gitignore !== false;
208
+ return JSON.stringify([call.name, "pattern", pattern, pathList, skip, caseInsensitive, gitignore]);
209
+ }
210
+ return undefined;
211
+ }
212
+
213
+ /**
214
+ * Files actually mutated according to a tool result's details. Used for
215
+ * AST-edit-shaped results (`ast_edit` direct-apply and the hidden `resolve`
216
+ * apply step), which report `{ applied: true, files: [...] }` — the resolve
217
+ * tool nests that payload under `details.sourceResultDetails`. Conservative:
218
+ * returns nothing unless the details explicitly mark the change as applied.
219
+ * Checked even on `isError` results: a stale-preview apply reports an error
220
+ * while still having mutated the listed files.
221
+ */
222
+ function resultDetailFiles(message: ToolResultMessage): string[] {
223
+ const raw = message.details as { applied?: unknown; files?: unknown; sourceResultDetails?: unknown } | undefined;
224
+ const candidates = [raw, raw?.sourceResultDetails as { applied?: unknown; files?: unknown } | undefined];
225
+ for (const details of candidates) {
226
+ if (details?.applied === true && Array.isArray(details.files)) {
227
+ return details.files.filter((file): file is string => typeof file === "string" && file.length > 0);
228
+ }
229
+ }
230
+ return [];
231
+ }
232
+
233
+ /**
234
+ * Paths that FAILED in a per-file edit result (`details.perFileResults`) and
235
+ * were NOT mutated by any same-path entry. Multi-file apply_patch catches
236
+ * per-file failures and still returns a non-error result; a purely-failed
237
+ * path was not mutated and must not stale reads. But apply_patch can emit
238
+ * multiple entries for the same path (e.g. several hunks): if any same-path
239
+ * entry succeeded the file still mutated, so it must NOT be suppressed.
240
+ * Conservative: only an entry explicitly marked `isError === true` counts as
241
+ * a failure; anything else (including ambiguous/malformed entries) counts as
242
+ * a success and keeps the path out of the suppression set.
243
+ */
244
+ function failedEditPaths(message: ToolResultMessage): Set<string> {
245
+ const details = message.details as { perFileResults?: unknown } | undefined;
246
+ const perFile = details?.perFileResults;
247
+ if (!Array.isArray(perFile)) return new Set();
248
+ const failed = new Set<string>();
249
+ const succeeded = new Set<string>();
250
+ for (const item of perFile) {
251
+ const entry = item as { path?: unknown; isError?: unknown };
252
+ if (typeof entry?.path !== "string") continue;
253
+ if (entry.isError === true) failed.add(entry.path);
254
+ else succeeded.add(entry.path);
255
+ }
256
+ // A path mutated if any same-path entry succeeded, even when another
257
+ // same-path entry failed; drop those from the suppression set.
258
+ for (const path of succeeded) failed.delete(path);
259
+ return failed;
260
+ }
261
+
262
+ /**
263
+ * Concrete file path a `read` result actually came from, when the tool
264
+ * reported one (`details.resolvedPath`). Suffix resolution can map a bare
265
+ * filename argument onto a different concrete path.
266
+ */
267
+ function readResolvedPath(message: ToolResultMessage): string | undefined {
268
+ const details = message.details as { resolvedPath?: unknown } | undefined;
269
+ const resolved = details?.resolvedPath;
270
+ return typeof resolved === "string" && resolved.length > 0 ? resolved : undefined;
271
+ }
272
+
273
+ interface StalenessIndex {
274
+ /** Entry indices of toolResults superseded by a later same-target result or a later edit. */
275
+ staleResultIndices: Set<number>;
276
+ }
277
+
278
+ /**
279
+ * Build a staleness index over session entries (oldest -> newest):
280
+ * - a toolResult is stale when a later non-error toolResult shares its target key;
281
+ * - a `read` result is stale when a later non-error edit/write touches its file.
282
+ * The most recent result per target is never stale.
283
+ */
284
+ function buildStalenessIndex(entries: SessionEntry[]): StalenessIndex {
285
+ const callsById = new Map<string, ToolCall>();
286
+ for (const entry of entries) {
287
+ if (entry.type !== "message") continue;
288
+ const message = entry.message as AgentMessage;
289
+ if (message.role !== "assistant") continue;
290
+ for (const content of message.content) {
291
+ if (content.type === "toolCall") callsById.set(content.id, content);
292
+ }
293
+ }
294
+
295
+ const lastResultIndexByKey = new Map<string, number>();
296
+ const resultMeta = new Map<number, { key?: string; call: ToolCall; message: ToolResultMessage }>();
297
+ const lastEditIndexByPath = new Map<string, number>();
298
+
299
+ for (let i = 0; i < entries.length; i++) {
300
+ const message = getToolResultMessage(entries[i]);
301
+ if (!message) continue;
302
+ const call = callsById.get(message.toolCallId);
303
+ if (!call) continue;
304
+
305
+ // AST edits mutate files when previews are applied via the hidden
306
+ // `resolve` tool; the call args carry globs, not concrete paths. Both
307
+ // tools report actually-touched files in result details. Collected
308
+ // BEFORE the error gate: a stale-preview apply reports an error while
309
+ // still having mutated the listed files.
310
+ if (call.name === "resolve" || call.name === "ast_edit") {
311
+ for (const editPath of resultDetailFiles(message)) {
312
+ lastEditIndexByPath.set(editPath, i);
313
+ }
314
+ }
315
+ if (message.isError) continue;
316
+
317
+ const key = toolTargetKey(call);
318
+ resultMeta.set(i, { key, call, message });
319
+ if (key !== undefined) lastResultIndexByKey.set(key, i);
320
+ if (EDIT_TOOL_NAMES.has(call.name)) {
321
+ // Per-file edit results record failures in details.perFileResults;
322
+ // a failed hunk mutated nothing, so exclude its whole path group
323
+ // (rename destination included) from touched paths.
324
+ const failed = failedEditPaths(message);
325
+ for (const group of editToolPathGroups(call)) {
326
+ if (group.some(groupPath => failed.has(groupPath))) continue;
327
+ for (const editPath of group) {
328
+ lastEditIndexByPath.set(editPath, i);
329
+ }
330
+ }
331
+ }
332
+ }
333
+
334
+ const staleResultIndices = new Set<number>();
335
+ for (const [index, meta] of resultMeta) {
336
+ if (meta.key !== undefined) {
337
+ const lastIndex = lastResultIndexByKey.get(meta.key);
338
+ if (lastIndex !== undefined && lastIndex > index) {
339
+ staleResultIndices.add(index);
340
+ continue;
341
+ }
342
+ }
343
+ if (meta.call.name === "read") {
344
+ // Check both the call argument (selectors stripped) and the resolved
345
+ // path from result details: suffix resolution can map a bare filename
346
+ // onto a different concrete path, and edits may use either form.
347
+ const lookupPaths = new Set<string>();
348
+ const argPath = toolCallPath(meta.call);
349
+ if (argPath !== undefined) lookupPaths.add(readBasePath(argPath));
350
+ const resolved = readResolvedPath(meta.message);
351
+ if (resolved !== undefined) lookupPaths.add(resolved);
352
+ for (const lookupPath of lookupPaths) {
353
+ const editIndex = lastEditIndexByPath.get(lookupPath);
354
+ if (editIndex !== undefined && editIndex > index) {
355
+ staleResultIndices.add(index);
356
+ break;
357
+ }
358
+ }
359
+ }
360
+ }
361
+
362
+ return { staleResultIndices };
363
+ }
364
+
46
365
  export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
47
366
  let accumulatedTokens = 0;
48
367
  let tokensSaved = 0;
49
368
  let prunedCount = 0;
50
369
 
51
- const candidates: Array<{ entry: SessionMessageEntry; tokens: number }> = [];
370
+ const { staleResultIndices } = buildStalenessIndex(entries);
371
+ const staleOverridable = new Set(config.staleOverridableTools ?? []);
372
+ const candidates: Array<{ entry: SessionMessageEntry; tokens: number; notice: string; savings: number }> = [];
52
373
 
53
374
  for (let i = entries.length - 1; i >= 0; i--) {
54
375
  const entry = entries[i];
55
376
  const message = getToolResultMessage(entry);
56
377
  if (!message) continue;
57
378
 
58
- const tokens = estimateTokens(message as AgentMessage);
59
- const isProtected = config.protectedTools.includes(message.toolName);
379
+ const tokens = estimateEntryTokens(entry);
380
+ const isStale = staleResultIndices.has(i);
381
+ // Staleness waives protected-tool immunity for overridable tools
382
+ // (e.g. a superseded `read`); the most recent result per target is
383
+ // never stale, so the latest read of each file stays protected.
384
+ const isProtected =
385
+ config.protectedTools.includes(message.toolName) && !(isStale && staleOverridable.has(message.toolName));
60
386
 
61
387
  if (message.prunedAt !== undefined) {
62
388
  accumulatedTokens += tokens;
63
389
  continue;
64
390
  }
65
391
 
66
- if (accumulatedTokens < config.protectTokens || isProtected) {
392
+ // Stale results are prunable even inside the recency protect window —
393
+ // they are superseded, so recency no longer implies relevance. They
394
+ // still count toward window accounting so non-stale protection is
395
+ // unchanged.
396
+ const insideProtectWindow = accumulatedTokens < config.protectTokens;
397
+ if ((insideProtectWindow && !isStale) || isProtected) {
67
398
  accumulatedTokens += tokens;
68
399
  continue;
69
400
  }
70
401
 
71
- candidates.push({ entry: entry as SessionMessageEntry, tokens });
402
+ const notice = createPrunedNotice(tokens, message);
403
+ candidates.push({
404
+ entry: entry as SessionMessageEntry,
405
+ tokens,
406
+ notice,
407
+ savings: estimatePrunedSavings(tokens, notice),
408
+ });
72
409
  accumulatedTokens += tokens;
73
410
  }
74
411
 
75
412
  for (const candidate of candidates) {
76
- tokensSaved += estimatePrunedSavings(candidate.tokens);
413
+ tokensSaved += candidate.savings;
77
414
  }
78
415
 
79
416
  if (tokensSaved < config.minimumSavings || candidates.length === 0) {
80
- return { prunedCount: 0, tokensSaved: 0 };
417
+ return { prunedCount: 0, tokensSaved: 0, prunedEntries: [] };
81
418
  }
82
419
 
83
420
  const prunedAt = Date.now();
421
+ const prunedEntries: SessionMessageEntry[] = [];
84
422
  for (const candidate of candidates) {
85
423
  const message = candidate.entry.message as ToolResultMessage;
86
- message.content = [{ type: "text", text: createPrunedNotice(candidate.tokens) }];
424
+ message.content = [{ type: "text", text: candidate.notice }];
87
425
  message.prunedAt = prunedAt;
426
+ prunedEntries.push(candidate.entry);
88
427
  prunedCount++;
89
428
  }
90
429
 
91
- return { prunedCount, tokensSaved };
430
+ return { prunedCount, tokensSaved, prunedEntries };
92
431
  }
package/src/types.ts CHANGED
@@ -189,6 +189,9 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
189
189
  */
190
190
  onAssistantMessageEvent?: (message: AssistantMessage, event: AssistantMessageEvent) => void;
191
191
 
192
+ /** Called for non-content tool-choice incapability stream events. */
193
+ onToolChoiceIncapability?: (event: Extract<AssistantMessageEvent, { type: "toolChoiceIncapability" }>) => void;
194
+
192
195
  /**
193
196
  * Called when GPT-5 Harmony protocol leakage is detected and mitigated.
194
197
  */