@librechat/agents 3.7.15 → 3.7.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/dist/cjs/agents/AgentContext.cjs +91 -0
  2. package/dist/cjs/agents/AgentContext.cjs.map +1 -1
  3. package/dist/cjs/agents/projection.cjs +2 -1
  4. package/dist/cjs/agents/projection.cjs.map +1 -1
  5. package/dist/cjs/graphs/Graph.cjs +47 -6
  6. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  7. package/dist/cjs/graphs/MultiAgentGraph.cjs +20 -7
  8. package/dist/cjs/graphs/MultiAgentGraph.cjs.map +1 -1
  9. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs +1 -1
  10. package/dist/cjs/llm/openai/utils/index.cjs +1 -1
  11. package/dist/cjs/llm/openai/utils/index.cjs.map +1 -1
  12. package/dist/cjs/llm/preempt.cjs +1 -10
  13. package/dist/cjs/llm/preempt.cjs.map +1 -1
  14. package/dist/cjs/main.cjs +23 -3
  15. package/dist/cjs/messages/contextPruning.cjs +1 -1
  16. package/dist/cjs/messages/contextPruning.cjs.map +1 -1
  17. package/dist/cjs/messages/core.cjs +29 -1
  18. package/dist/cjs/messages/core.cjs.map +1 -1
  19. package/dist/cjs/messages/fading.cjs +122 -0
  20. package/dist/cjs/messages/fading.cjs.map +1 -0
  21. package/dist/cjs/messages/format.cjs +1 -1
  22. package/dist/cjs/messages/index.cjs +1 -0
  23. package/dist/cjs/messages/prune.cjs +364 -236
  24. package/dist/cjs/messages/prune.cjs.map +1 -1
  25. package/dist/cjs/run.cjs +27 -0
  26. package/dist/cjs/run.cjs.map +1 -1
  27. package/dist/cjs/session/AgentSession.cjs +213 -8
  28. package/dist/cjs/session/AgentSession.cjs.map +1 -1
  29. package/dist/cjs/session/JsonlSessionStore.cjs +41 -0
  30. package/dist/cjs/session/JsonlSessionStore.cjs.map +1 -1
  31. package/dist/cjs/session/index.cjs +1 -1
  32. package/dist/cjs/stream.cjs +1 -1
  33. package/dist/cjs/summarization/node.cjs +1 -1
  34. package/dist/cjs/tools/ToolNode.cjs +1 -1
  35. package/dist/cjs/tools/subagent/SubagentExecutor.cjs +2 -0
  36. package/dist/cjs/tools/subagent/SubagentExecutor.cjs.map +1 -1
  37. package/dist/cjs/tools/subagent/SubagentReplay.cjs +2 -1
  38. package/dist/cjs/tools/subagent/SubagentReplay.cjs.map +1 -1
  39. package/dist/cjs/utils/errors.cjs +3 -1
  40. package/dist/cjs/utils/errors.cjs.map +1 -1
  41. package/dist/cjs/utils/index.cjs +1 -1
  42. package/dist/cjs/utils/truncation.cjs +9 -0
  43. package/dist/cjs/utils/truncation.cjs.map +1 -1
  44. package/dist/esm/agents/AgentContext.mjs +92 -1
  45. package/dist/esm/agents/AgentContext.mjs.map +1 -1
  46. package/dist/esm/agents/projection.mjs +2 -1
  47. package/dist/esm/agents/projection.mjs.map +1 -1
  48. package/dist/esm/graphs/Graph.mjs +47 -6
  49. package/dist/esm/graphs/Graph.mjs.map +1 -1
  50. package/dist/esm/graphs/MultiAgentGraph.mjs +20 -7
  51. package/dist/esm/graphs/MultiAgentGraph.mjs.map +1 -1
  52. package/dist/esm/llm/anthropic/utils/message_inputs.mjs +1 -1
  53. package/dist/esm/llm/openai/utils/index.mjs +2 -2
  54. package/dist/esm/llm/openai/utils/index.mjs.map +1 -1
  55. package/dist/esm/llm/preempt.mjs +1 -10
  56. package/dist/esm/llm/preempt.mjs.map +1 -1
  57. package/dist/esm/main.mjs +6 -5
  58. package/dist/esm/messages/contextPruning.mjs +1 -1
  59. package/dist/esm/messages/contextPruning.mjs.map +1 -1
  60. package/dist/esm/messages/core.mjs +29 -2
  61. package/dist/esm/messages/core.mjs.map +1 -1
  62. package/dist/esm/messages/fading.mjs +108 -0
  63. package/dist/esm/messages/fading.mjs.map +1 -0
  64. package/dist/esm/messages/format.mjs +1 -1
  65. package/dist/esm/messages/index.mjs +1 -0
  66. package/dist/esm/messages/prune.mjs +363 -235
  67. package/dist/esm/messages/prune.mjs.map +1 -1
  68. package/dist/esm/run.mjs +27 -0
  69. package/dist/esm/run.mjs.map +1 -1
  70. package/dist/esm/session/AgentSession.mjs +213 -8
  71. package/dist/esm/session/AgentSession.mjs.map +1 -1
  72. package/dist/esm/session/JsonlSessionStore.mjs +41 -0
  73. package/dist/esm/session/JsonlSessionStore.mjs.map +1 -1
  74. package/dist/esm/session/index.mjs +1 -1
  75. package/dist/esm/stream.mjs +1 -1
  76. package/dist/esm/summarization/node.mjs +1 -1
  77. package/dist/esm/tools/ToolNode.mjs +1 -1
  78. package/dist/esm/tools/subagent/SubagentExecutor.mjs +2 -0
  79. package/dist/esm/tools/subagent/SubagentExecutor.mjs.map +1 -1
  80. package/dist/esm/tools/subagent/SubagentReplay.mjs +2 -1
  81. package/dist/esm/tools/subagent/SubagentReplay.mjs.map +1 -1
  82. package/dist/esm/utils/errors.mjs +3 -1
  83. package/dist/esm/utils/errors.mjs.map +1 -1
  84. package/dist/esm/utils/index.mjs +1 -1
  85. package/dist/esm/utils/truncation.mjs +7 -1
  86. package/dist/esm/utils/truncation.mjs.map +1 -1
  87. package/dist/types/agents/AgentContext.d.ts +32 -1
  88. package/dist/types/agents/projection.d.ts +3 -1
  89. package/dist/types/graphs/Graph.d.ts +14 -1
  90. package/dist/types/graphs/MultiAgentGraph.d.ts +7 -0
  91. package/dist/types/messages/contextPruning.d.ts +2 -0
  92. package/dist/types/messages/core.d.ts +2 -0
  93. package/dist/types/messages/fading.d.ts +81 -0
  94. package/dist/types/messages/index.d.ts +1 -0
  95. package/dist/types/messages/prune.d.ts +71 -24
  96. package/dist/types/run.d.ts +16 -0
  97. package/dist/types/session/AgentSession.d.ts +15 -0
  98. package/dist/types/session/JsonlSessionStore.d.ts +4 -1
  99. package/dist/types/session/types.d.ts +9 -0
  100. package/dist/types/tools/subagent/SubagentReplay.d.ts +3 -1
  101. package/dist/types/types/graph.d.ts +23 -0
  102. package/dist/types/types/run.d.ts +10 -0
  103. package/dist/types/utils/truncation.d.ts +9 -0
  104. package/package.json +1 -1
  105. package/src/agents/AgentContext.ts +175 -1
  106. package/src/agents/projection.ts +4 -0
  107. package/src/graphs/Graph.ts +97 -11
  108. package/src/graphs/MultiAgentGraph.ts +54 -6
  109. package/src/llm/openai/utils/index.ts +3 -1
  110. package/src/llm/preempt.ts +4 -17
  111. package/src/messages/contextPruning.ts +7 -1
  112. package/src/messages/core.ts +52 -0
  113. package/src/messages/fading.ts +301 -0
  114. package/src/messages/index.ts +1 -0
  115. package/src/messages/prune.ts +719 -507
  116. package/src/run.ts +52 -0
  117. package/src/session/AgentSession.ts +397 -8
  118. package/src/session/JsonlSessionStore.ts +71 -0
  119. package/src/session/types.ts +10 -0
  120. package/src/tools/subagent/SubagentExecutor.ts +6 -0
  121. package/src/tools/subagent/SubagentReplay.ts +15 -2
  122. package/src/types/graph.ts +25 -0
  123. package/src/types/run.ts +10 -0
  124. package/src/utils/errors.ts +17 -1
  125. package/src/utils/truncation.ts +25 -0
@@ -11,29 +11,43 @@ import type {
11
11
  MessageContentComplex,
12
12
  ReasoningContentText,
13
13
  } from '@/types/stream';
14
- import type { ContextPruningConfig } from '@/types/graph';
14
+ import type { ContextPruningConfig, FadingTier } from '@/types/graph';
15
+ import type { FadingCaps, FadingSignals } from './fading';
15
16
  import type { TokenCounter } from '@/types/run';
16
17
  import type { ProviderName } from '@/types';
18
+ import {
19
+ HARD_MAX_TOOL_CALL_INPUT_CHARS,
20
+ HARD_MAX_TOOL_RESULT_CHARS,
21
+ MIN_JSON_VALUE_CHARS,
22
+ calculateMaxToolCallInputChars,
23
+ calculateMaxToolResultChars,
24
+ } from '@/utils/truncation';
17
25
  import {
18
26
  cloneToolMessageWithContent,
19
27
  compactToolContent,
20
- getToolContentCharLength,
21
28
  isComputerCallOutputMessage,
22
29
  serializeStructuredValueBounded,
23
30
  serializeToolContentBounded,
24
31
  } from '@/utils/toolContent';
25
32
  import {
26
- HARD_MAX_TOOL_RESULT_CHARS,
27
- calculateMaxToolResultChars,
28
- truncateToolResultContent,
29
- } from '@/utils/truncation';
33
+ MASKED_RESULT_MIN_CHARS,
34
+ fadingRungForExchangeChars,
35
+ isFadingTier,
36
+ resolveFadingCaps,
37
+ resolveFadingTier,
38
+ seedFadingTier,
39
+ } from './fading';
40
+ import {
41
+ dropIncompleteToolStreamContent,
42
+ hasNonEmptyTextContent,
43
+ } from './core';
30
44
  import { resolveContextPruningSettings } from './contextPruningSettings';
31
45
  import { hasUnsafeStructuredSerialization } from '@/utils/tokens';
32
46
  import { ContentTypes, Providers, Constants } from '@/common';
33
47
  import { getProviderFamily } from '@/llm/providerRegistry';
34
- import { dropIncompleteToolStreamContent } from './core';
35
48
  import { applyContextPruning } from './contextPruning';
36
49
  import { toLangChainContent } from './langchain';
50
+ import { cloneMessage } from './cache';
37
51
 
38
52
  function sumTokenCounts(
39
53
  tokenMap: Record<string, number | undefined>,
@@ -74,20 +88,6 @@ export const DEFAULT_RESERVE_RATIO = 0.05;
74
88
  /** Provider framing reserved for the assistant reply label. */
75
89
  export const REPLY_PRIMER_TOKENS = 3;
76
90
 
77
- /** Context pressure at which observation masking and context fading activate. */
78
- const PRESSURE_THRESHOLD_MASKING = 0.8;
79
-
80
- /** Pressure band thresholds paired with budget factors for progressive context fading. */
81
- const PRESSURE_BANDS: [number, number][] = [
82
- [0.99, 0.05],
83
- [0.9, 0.2],
84
- [0.85, 0.5],
85
- [0.8, 1.0],
86
- ];
87
-
88
- /** Maximum character length for masked (consumed) tool results. */
89
- const MASKED_RESULT_MAX_CHARS = 300;
90
-
91
91
  /** Hard cap for the originalToolContent store (~2 MB estimated from char length). */
92
92
  export const ORIGINAL_CONTENT_MAX_CHARS = 2_000_000;
93
93
 
@@ -172,6 +172,13 @@ export type PruneMessagesFactoryParams = {
172
172
  * of waiting for the first provider response. Ignored when <= 0.
173
173
  */
174
174
  calibrationRatio?: number;
175
+ /**
176
+ * Context-fading tier persisted from a previous run's contextMeta. Seeds the
177
+ * latched cap ladder so historical tool results keep the same truncated
178
+ * bytes across runs. Invalid values start fresh; valid values clamp to the
179
+ * current context window without losing their latched provenance.
180
+ */
181
+ fadingTier?: FadingTier | null;
175
182
  /** Optional diagnostic log callback wired by the graph for observability. */
176
183
  log?: (
177
184
  level: 'debug' | 'info' | 'warn' | 'error',
@@ -180,6 +187,18 @@ export type PruneMessagesFactoryParams = {
180
187
  ) => void;
181
188
  };
182
189
  export type PruneMessagesParams = {
190
+ /**
191
+ * Immutable graph history corresponding index-for-index with `messages`.
192
+ * When supplied, provider projections always derive from this source rather
193
+ * than from an earlier, already-truncated projection.
194
+ */
195
+ canonicalMessages?: BaseMessage[];
196
+ /**
197
+ * The caller guarantees that an existing canonical prefix cannot have been
198
+ * rewritten since the previous call. Graph reducers provide this guarantee
199
+ * by invalidating and recreating the pruner on replacements/removals.
200
+ */
201
+ canonicalPrefixStable?: boolean;
183
202
  messages: BaseMessage[];
184
203
  usageMetadata?: Partial<UsageMetadata>;
185
204
  startType?: ReturnType<BaseMessage['getType']>;
@@ -209,24 +228,52 @@ function getToolCallIds(message: BaseMessage): Set<string> {
209
228
 
210
229
  const ids = new Set<string>();
211
230
  const aiMessage = message as AIMessage;
212
- for (const toolCall of aiMessage.tool_calls ?? []) {
213
- if (typeof toolCall.id === 'string' && toolCall.id.length > 0) {
214
- ids.add(toolCall.id);
231
+ const toolCalls = aiMessage.tool_calls;
232
+ if (Array.isArray(toolCalls) && !isProxy(toolCalls)) {
233
+ for (const toolCall of toolCalls) {
234
+ if (typeof toolCall !== 'object') {
235
+ continue;
236
+ }
237
+ const id = getStringProperty(toolCall, 'id');
238
+ if (id != null && id.length > 0) {
239
+ ids.add(id);
240
+ }
215
241
  }
216
242
  }
217
243
 
218
- if (Array.isArray(aiMessage.content)) {
244
+ const rawToolCalls = readPropertyWithoutAccessors(
245
+ aiMessage.additional_kwargs,
246
+ 'tool_calls'
247
+ );
248
+ if (
249
+ !rawToolCalls.accessor &&
250
+ Array.isArray(rawToolCalls.value) &&
251
+ !isProxy(rawToolCalls.value)
252
+ ) {
253
+ for (const toolCall of rawToolCalls.value) {
254
+ if (toolCall == null || typeof toolCall !== 'object') {
255
+ continue;
256
+ }
257
+ const id = getStringProperty(toolCall, 'id');
258
+ if (id != null && id.length > 0) {
259
+ ids.add(id);
260
+ }
261
+ }
262
+ }
263
+
264
+ if (Array.isArray(aiMessage.content) && !isProxy(aiMessage.content)) {
219
265
  for (const part of aiMessage.content) {
220
266
  if (typeof part !== 'object') {
221
267
  continue;
222
268
  }
223
- const record = part as { type?: unknown; id?: unknown };
269
+ const type = getStringProperty(part, 'type');
270
+ const id = getStringProperty(part, 'id');
224
271
  if (
225
- (record.type === 'tool_use' || record.type === 'tool_call') &&
226
- typeof record.id === 'string' &&
227
- record.id.length > 0
272
+ (type === 'tool_use' || type === 'tool_call') &&
273
+ id != null &&
274
+ id.length > 0
228
275
  ) {
229
- ids.add(record.id);
276
+ ids.add(id);
230
277
  }
231
278
  }
232
279
  }
@@ -237,18 +284,16 @@ function getToolCallIds(message: BaseMessage): Set<string> {
237
284
  if (item == null || typeof item !== 'object') {
238
285
  continue;
239
286
  }
240
- const record = item as {
241
- type?: unknown;
242
- call_id?: unknown;
243
- };
287
+ const type = getStringProperty(item, 'type');
288
+ const callId = getStringProperty(item, 'call_id');
244
289
  if (
245
- (record.type === 'function_call' ||
246
- record.type === 'custom_tool_call' ||
247
- record.type === 'computer_call') &&
248
- typeof record.call_id === 'string' &&
249
- record.call_id !== ''
290
+ (type === 'function_call' ||
291
+ type === 'custom_tool_call' ||
292
+ type === 'computer_call') &&
293
+ callId != null &&
294
+ callId !== ''
250
295
  ) {
251
- ids.add(record.call_id);
296
+ ids.add(callId);
252
297
  }
253
298
  }
254
299
  }
@@ -1135,139 +1180,224 @@ export function checkValidNumber(value: unknown): value is number {
1135
1180
  return typeof value === 'number' && !isNaN(value) && value > 0;
1136
1181
  }
1137
1182
 
1138
- /**
1139
- * Observation masking: replaces consumed ToolMessage content with tight
1140
- * head+tail truncations that serve as informative placeholders.
1141
- *
1142
- * A ToolMessage is "consumed" when a subsequent AI message exists that is NOT
1143
- * purely tool calls — meaning the model has already read and acted on the
1144
- * result. Unconsumed results (the latest tool outputs the model hasn't
1145
- * responded to yet) are left intact so the model can still use them.
1146
- *
1147
- * AI messages are never masked — they contain the model's own reasoning and
1148
- * conclusions, which is what prevents the model from repeating work after
1149
- * its tool results are masked.
1150
- *
1151
- * @returns The number of tool messages that were masked.
1152
- */
1153
- export function maskConsumedToolResults(params: {
1183
+ type FadingApplyParams = {
1184
+ canonicalMessages?: BaseMessage[];
1154
1185
  messages: BaseMessage[];
1155
1186
  indexTokenCountMap: Record<string, number | undefined>;
1156
1187
  tokenCounter: TokenCounter;
1157
- /** Raw-space token budget available for all consumed tool results combined.
1158
- * When provided, the budget is distributed across consumed results weighted
1159
- * by recency (newest get the most, oldest get MASKED_RESULT_MAX_CHARS min).
1160
- * When omitted, falls back to a flat MASKED_RESULT_MAX_CHARS per result. */
1161
- availableRawBudget?: number;
1162
- /** When provided, original (pre-masking) content is stored here keyed by
1163
- * message index — only for entries that actually get truncated. */
1188
+ caps: Pick<FadingCaps, 'resultChars' | 'consumedChars' | 'inputChars'>;
1189
+ /** Whether consumed results shrink to `caps.consumedChars`. */
1190
+ masked: boolean;
1191
+ /** First index to visit for fresh results and tool-call inputs. */
1192
+ fromIndex?: number;
1193
+ /** First consumed index to visit for masking. */
1194
+ maskedFromIndex?: number;
1195
+ /** Original (pre-masking) content keyed by message index, captured for the summarizer. */
1164
1196
  originalContentStore?: Map<number, string>;
1165
1197
  /** Called after storing a newly captured entry. */
1166
1198
  onContentStored?: (index: number, content: string) => void;
1167
- }): number {
1168
- const { messages, indexTokenCountMap, tokenCounter } = params;
1169
- let maskedCount = 0;
1199
+ };
1170
1200
 
1171
- // Pass 1 (backward): identify consumed tool message indices.
1172
- // A ToolMessage is "consumed" once we've seen a subsequent AI message with
1173
- // substantive text content (not just tool calls).
1174
- // Collected in forward order (oldest first) for recency weighting.
1175
- let seenNonToolCallAI = false;
1176
- const consumedIndices: number[] = [];
1201
+ export type FadingApplyResult = {
1202
+ /** Fresh tool results rewritten. */
1203
+ truncated: number;
1204
+ /** Tool-call inputs rewritten. */
1205
+ inputs: number;
1206
+ /** Consumed tool results rewritten. */
1207
+ masked: number;
1208
+ /** Index of the newest AI message with text; tool results before it are consumed. */
1209
+ consumedBoundary: number;
1210
+ };
1177
1211
 
1212
+ /**
1213
+ * Index of the newest AI message with substantive text. Every ToolMessage
1214
+ * before it has been answered by the model ("consumed"); results after it are
1215
+ * still fresh. Scans backward, so the cost is bounded by the last turn.
1216
+ */
1217
+ function findConsumedBoundary(messages: BaseMessage[]): number {
1178
1218
  for (let i = messages.length - 1; i >= 0; i--) {
1179
- const msg = messages[i];
1180
- const type = msg.getType();
1181
-
1182
- if (type === 'ai') {
1183
- const hasText =
1184
- typeof msg.content === 'string'
1185
- ? msg.content.trim().length > 0
1186
- : Array.isArray(msg.content) &&
1187
- msg.content.some(
1188
- (b) =>
1189
- typeof b === 'object' &&
1190
- (b as Record<string, unknown>).type === 'text' &&
1191
- typeof (b as Record<string, unknown>).text === 'string' &&
1192
- ((b as Record<string, unknown>).text as string).trim().length >
1193
- 0
1194
- );
1195
- if (hasText) {
1196
- seenNonToolCallAI = true;
1197
- }
1198
- } else if (type === 'tool' && seenNonToolCallAI) {
1199
- consumedIndices.push(i);
1219
+ const message = messages[i];
1220
+ if (message.getType() === 'ai' && hasNonEmptyTextContent(message.content)) {
1221
+ return i;
1200
1222
  }
1201
1223
  }
1224
+ return 0;
1225
+ }
1202
1226
 
1203
- if (consumedIndices.length === 0) {
1204
- return 0;
1205
- }
1206
-
1207
- consumedIndices.reverse();
1208
-
1209
- const totalBudgetChars =
1210
- params.availableRawBudget != null && params.availableRawBudget > 0
1211
- ? params.availableRawBudget * 4
1212
- : 0;
1213
-
1214
- const count = consumedIndices.length;
1227
+ /**
1228
+ * Applies a fading tier's caps in one forward pass. Consumed results (before
1229
+ * the boundary) shrink to `consumedChars`, fresh results to `resultChars` and
1230
+ * historical tool-call inputs to `inputChars`. Messages already within their
1231
+ * cap keep object identity and token count, so at an unchanged tier the pass
1232
+ * only touches what arrived since the watermarks. Truncation is a pure
1233
+ * function of (content, cap), which is what keeps the bytes of a historical
1234
+ * result identical from call to call.
1235
+ */
1236
+ export function applyFadingCaps(params: FadingApplyParams): FadingApplyResult {
1237
+ const { messages, indexTokenCountMap, tokenCounter, caps, masked } = params;
1238
+ const fromIndex = params.fromIndex ?? 0;
1239
+ const maskedFromIndex = params.maskedFromIndex ?? 0;
1240
+ const consumedBoundary = masked ? findConsumedBoundary(messages) : 0;
1241
+ const start = masked ? Math.min(fromIndex, maskedFromIndex) : fromIndex;
1242
+ let truncated = 0;
1243
+ let maskedCount = 0;
1215
1244
 
1216
- for (let c = 0; c < count; c++) {
1217
- const i = consumedIndices[c];
1245
+ for (let i = start; i < messages.length; i++) {
1218
1246
  const message = messages[i];
1219
- if (isComputerCallOutputMessage(message)) {
1247
+ if (message.getType() !== 'tool' || isComputerCallOutputMessage(message)) {
1220
1248
  continue;
1221
1249
  }
1222
- const content = message.content;
1223
-
1224
- let maxChars: number;
1225
- if (totalBudgetChars > 0) {
1226
- const position = count > 1 ? c / (count - 1) : 1;
1227
- const weight = 0.2 + 0.8 * position;
1228
- const totalWeight = count > 1 ? 0.6 * count : 1;
1229
- const share = (weight / totalWeight) * totalBudgetChars;
1230
- maxChars = Math.max(MASKED_RESULT_MAX_CHARS, Math.floor(share));
1231
- } else {
1232
- maxChars = MASKED_RESULT_MAX_CHARS;
1250
+ const consumed = masked && i < consumedBoundary;
1251
+ if (consumed ? i < maskedFromIndex : i < fromIndex) {
1252
+ continue;
1233
1253
  }
1234
-
1235
- const compacted = compactToolContent(content, maxChars);
1254
+ const maxChars = consumed ? caps.consumedChars : caps.resultChars;
1255
+ if (!Number.isFinite(maxChars)) {
1256
+ continue;
1257
+ }
1258
+ const canonicalContent =
1259
+ params.canonicalMessages?.[i]?.content ?? message.content;
1260
+ const compacted = compactToolContent(canonicalContent, maxChars);
1236
1261
  if (!compacted.changed) {
1237
1262
  continue;
1238
1263
  }
1239
-
1240
- if (params.originalContentStore && !params.originalContentStore.has(i)) {
1264
+ if (
1265
+ consumed &&
1266
+ params.originalContentStore &&
1267
+ !params.originalContentStore.has(i)
1268
+ ) {
1241
1269
  const original = serializeToolContentBounded(
1242
- content,
1270
+ canonicalContent,
1243
1271
  ORIGINAL_CONTENT_MAX_CHARS
1244
1272
  );
1245
1273
  params.originalContentStore.set(i, original);
1246
- if (params.onContentStored) {
1247
- params.onContentStored(i, original);
1248
- }
1274
+ params.onContentStored?.(i, original);
1249
1275
  }
1250
-
1251
1276
  const cloned = cloneToolMessageWithContent(
1252
1277
  message as ToolMessage,
1253
1278
  compacted.content
1254
1279
  );
1255
1280
  messages[i] = cloned;
1256
1281
  indexTokenCountMap[i] = tokenCounter(cloned);
1257
- maskedCount++;
1282
+ if (consumed) {
1283
+ maskedCount++;
1284
+ } else {
1285
+ truncated++;
1286
+ }
1258
1287
  }
1259
1288
 
1260
- return maskedCount;
1289
+ const inputs = Number.isFinite(caps.inputChars)
1290
+ ? applyToolCallInputCaps({
1291
+ messages,
1292
+ canonicalMessages: params.canonicalMessages,
1293
+ maxInputChars: caps.inputChars,
1294
+ indexTokenCountMap,
1295
+ tokenCounter,
1296
+ fromIndex,
1297
+ })
1298
+ : 0;
1299
+
1300
+ return { truncated, inputs, masked: maskedCount, consumedBoundary };
1261
1301
  }
1262
1302
 
1263
1303
  /**
1264
- * Pre-flight truncation: truncates oversized ToolMessage content before the
1265
- * main backward-iteration pruning runs. Unlike the ingestion guard (which caps
1266
- * at tool-execution time), pre-flight truncation applies per-turn based on the
1267
- * current context window budget (which may have shrunk due to growing conversation).
1304
+ * Observation masking: replaces consumed ToolMessage content with tight
1305
+ * head+tail truncations that serve as informative placeholders. Fresh results
1306
+ * and tool-call inputs are left alone.
1268
1307
  *
1269
- * After truncation, recounts tokens via tokenCounter and updates indexTokenCountMap
1270
- * so subsequent pruning works with accurate counts.
1308
+ * @returns The number of tool messages that were masked.
1309
+ */
1310
+ export function maskConsumedToolResults(params: {
1311
+ messages: BaseMessage[];
1312
+ indexTokenCountMap: Record<string, number | undefined>;
1313
+ tokenCounter: TokenCounter;
1314
+ /** Character cap applied to every consumed result (never below
1315
+ * MASKED_RESULT_MIN_CHARS, which is also the default). */
1316
+ maxChars?: number;
1317
+ /** @deprecated Aggregate raw-token budget distributed by recency. Prefer
1318
+ * `maxChars` for byte-stable masking across otherwise identical calls. */
1319
+ availableRawBudget?: number;
1320
+ /** When provided, original (pre-masking) content is stored here keyed by
1321
+ * message index — only for entries that actually get truncated. */
1322
+ originalContentStore?: Map<number, string>;
1323
+ /** Called after storing a newly captured entry. */
1324
+ onContentStored?: (index: number, content: string) => void;
1325
+ }): number {
1326
+ if (
1327
+ params.maxChars == null &&
1328
+ params.availableRawBudget != null &&
1329
+ params.availableRawBudget > 0
1330
+ ) {
1331
+ const consumedBoundary = findConsumedBoundary(params.messages);
1332
+ const consumedIndices: number[] = [];
1333
+ for (let i = 0; i < consumedBoundary; i++) {
1334
+ if (params.messages[i].getType() === 'tool') {
1335
+ consumedIndices.push(i);
1336
+ }
1337
+ }
1338
+ const count = consumedIndices.length;
1339
+ const totalBudgetChars = params.availableRawBudget * 4;
1340
+ let masked = 0;
1341
+ for (let c = 0; c < count; c++) {
1342
+ const i = consumedIndices[c];
1343
+ const message = params.messages[i];
1344
+ if (isComputerCallOutputMessage(message)) {
1345
+ continue;
1346
+ }
1347
+ const position = count > 1 ? c / (count - 1) : 1;
1348
+ const weight = 0.2 + 0.8 * position;
1349
+ const totalWeight = count > 1 ? 0.6 * count : 1;
1350
+ const maxChars = Math.max(
1351
+ MASKED_RESULT_MIN_CHARS,
1352
+ Math.floor((weight / totalWeight) * totalBudgetChars)
1353
+ );
1354
+ const compacted = compactToolContent(message.content, maxChars);
1355
+ if (!compacted.changed) {
1356
+ continue;
1357
+ }
1358
+ if (
1359
+ params.originalContentStore != null &&
1360
+ !params.originalContentStore.has(i)
1361
+ ) {
1362
+ const original = serializeToolContentBounded(
1363
+ message.content,
1364
+ ORIGINAL_CONTENT_MAX_CHARS
1365
+ );
1366
+ params.originalContentStore.set(i, original);
1367
+ params.onContentStored?.(i, original);
1368
+ }
1369
+ const cloned = cloneToolMessageWithContent(
1370
+ message as ToolMessage,
1371
+ compacted.content
1372
+ );
1373
+ params.messages[i] = cloned;
1374
+ params.indexTokenCountMap[i] = params.tokenCounter(cloned);
1375
+ masked++;
1376
+ }
1377
+ return masked;
1378
+ }
1379
+ return applyFadingCaps({
1380
+ messages: params.messages,
1381
+ indexTokenCountMap: params.indexTokenCountMap,
1382
+ tokenCounter: params.tokenCounter,
1383
+ caps: {
1384
+ resultChars: Number.POSITIVE_INFINITY,
1385
+ consumedChars: Math.max(
1386
+ MASKED_RESULT_MIN_CHARS,
1387
+ Math.floor(params.maxChars ?? MASKED_RESULT_MIN_CHARS)
1388
+ ),
1389
+ inputChars: Number.POSITIVE_INFINITY,
1390
+ },
1391
+ masked: true,
1392
+ originalContentStore: params.originalContentStore,
1393
+ onContentStored: params.onContentStored,
1394
+ }).masked;
1395
+ }
1396
+
1397
+ /**
1398
+ * Pre-flight truncation: truncates oversized ToolMessage content before the
1399
+ * main backward-iteration pruning runs, applying one cap derived from
1400
+ * `maxContextTokens` to every tool result.
1271
1401
  *
1272
1402
  * @returns The number of tool messages that were truncated.
1273
1403
  */
@@ -1277,44 +1407,17 @@ export function preFlightTruncateToolResults(params: {
1277
1407
  indexTokenCountMap: Record<string, number | undefined>;
1278
1408
  tokenCounter: TokenCounter;
1279
1409
  }): number {
1280
- const { messages, maxContextTokens, indexTokenCountMap, tokenCounter } =
1281
- params;
1282
- const baseMaxChars = calculateMaxToolResultChars(maxContextTokens);
1283
- let truncatedCount = 0;
1284
-
1285
- const toolIndices: number[] = [];
1286
- for (let i = 0; i < messages.length; i++) {
1287
- if (
1288
- messages[i].getType() === 'tool' &&
1289
- !isComputerCallOutputMessage(messages[i])
1290
- ) {
1291
- toolIndices.push(i);
1292
- }
1293
- }
1294
-
1295
- for (let t = 0; t < toolIndices.length; t++) {
1296
- const i = toolIndices[t];
1297
- const message = messages[i];
1298
- const content = message.content;
1299
-
1300
- const position = toolIndices.length > 1 ? t / (toolIndices.length - 1) : 1;
1301
- const recencyFactor = 0.2 + 0.8 * position;
1302
- const maxChars = Math.max(200, Math.floor(baseMaxChars * recencyFactor));
1303
-
1304
- const compacted = compactToolContent(content, maxChars);
1305
- if (!compacted.changed) {
1306
- continue;
1307
- }
1308
- const cloned = cloneToolMessageWithContent(
1309
- message as ToolMessage,
1310
- compacted.content
1311
- );
1312
- messages[i] = cloned;
1313
- indexTokenCountMap[i] = tokenCounter(cloned);
1314
- truncatedCount++;
1315
- }
1316
-
1317
- return truncatedCount;
1410
+ return applyFadingCaps({
1411
+ messages: params.messages,
1412
+ indexTokenCountMap: params.indexTokenCountMap,
1413
+ tokenCounter: params.tokenCounter,
1414
+ caps: {
1415
+ resultChars: calculateMaxToolResultChars(params.maxContextTokens),
1416
+ consumedChars: Number.POSITIVE_INFINITY,
1417
+ inputChars: Number.POSITIVE_INFINITY,
1418
+ },
1419
+ masked: false,
1420
+ }).truncated;
1318
1421
  }
1319
1422
 
1320
1423
  /**
@@ -1330,8 +1433,6 @@ export function preFlightTruncateToolResults(params: {
1330
1433
  *
1331
1434
  * @returns The number of AI messages that had tool_use inputs truncated.
1332
1435
  */
1333
- const HARD_MAX_TOOL_CALL_INPUT_CHARS = 200_000;
1334
- const MIN_JSON_VALUE_CHARS = 4;
1335
1436
  const ACCESSOR_INPUT_PLACEHOLDER = '[Property accessor omitted]';
1336
1437
 
1337
1438
  type ToolInputProjection = {
@@ -1454,14 +1555,19 @@ function cloneAIMessageWithProjectedStreamContent(
1454
1555
  ) as AIMessage | AIMessageChunk;
1455
1556
  }
1456
1557
 
1558
+ const TOOL_INPUT_TRUNCATION_MARKER = '… [truncated]\n';
1559
+
1457
1560
  function createBoundedTruncationValue(
1458
1561
  preview: string,
1459
1562
  originalChars: number,
1460
1563
  maxChars: number
1461
1564
  ): unknown {
1462
1565
  const normalizedMaxChars = normalizeToolInputLimit(maxChars);
1566
+ const canonicalPrefix = preview.startsWith(TOOL_INPUT_TRUNCATION_MARKER)
1567
+ ? preview.slice(TOOL_INPUT_TRUNCATION_MARKER.length)
1568
+ : preview;
1463
1569
  const emptyEnvelope = {
1464
- _truncated: '',
1570
+ _truncated: TOOL_INPUT_TRUNCATION_MARKER,
1465
1571
  _originalChars: originalChars,
1466
1572
  };
1467
1573
  if (JSON.stringify(emptyEnvelope).length > normalizedMaxChars) {
@@ -1480,11 +1586,12 @@ function createBoundedTruncationValue(
1480
1586
  }
1481
1587
 
1482
1588
  let low = 0;
1483
- let high = Math.min(preview.length, normalizedMaxChars);
1589
+ let high = Math.min(canonicalPrefix.length, normalizedMaxChars);
1484
1590
  while (low < high) {
1485
1591
  const next = Math.ceil((low + high) / 2);
1486
1592
  const candidate = {
1487
- _truncated: preview.slice(0, next),
1593
+ _truncated:
1594
+ TOOL_INPUT_TRUNCATION_MARKER + canonicalPrefix.slice(0, next),
1488
1595
  _originalChars: originalChars,
1489
1596
  };
1490
1597
  if (JSON.stringify(candidate).length <= normalizedMaxChars) {
@@ -1493,27 +1600,81 @@ function createBoundedTruncationValue(
1493
1600
  high = next - 1;
1494
1601
  }
1495
1602
  }
1496
- const truncationMarker = '… [truncated]';
1497
- const boundedPreview =
1498
- low < preview.length && low >= truncationMarker.length
1499
- ? preview.slice(0, low - truncationMarker.length) + truncationMarker
1500
- : preview.slice(0, low);
1501
1603
  return {
1502
- _truncated: boundedPreview,
1604
+ // Keep the marker separate from a pure canonical prefix so another,
1605
+ // slightly smaller cap can be derived without nesting the envelope.
1606
+ _truncated:
1607
+ TOOL_INPUT_TRUNCATION_MARKER + canonicalPrefix.slice(0, low),
1503
1608
  _originalChars: originalChars,
1504
1609
  };
1505
1610
  }
1506
1611
 
1612
+ function readBoundedTruncationValue(
1613
+ input: unknown
1614
+ ): { preview: string; originalChars: number } | undefined {
1615
+ if (input == null || typeof input !== 'object' || isProxy(input)) {
1616
+ return undefined;
1617
+ }
1618
+ try {
1619
+ const prototype = Object.getPrototypeOf(input);
1620
+ const keys = Object.keys(input);
1621
+ if (
1622
+ (prototype !== Object.prototype && prototype !== null) ||
1623
+ keys.length !== 2 ||
1624
+ !keys.includes('_truncated') ||
1625
+ !keys.includes('_originalChars')
1626
+ ) {
1627
+ return undefined;
1628
+ }
1629
+ } catch {
1630
+ return undefined;
1631
+ }
1632
+ const preview = readPropertyWithoutAccessors(input, '_truncated');
1633
+ const originalChars = readPropertyWithoutAccessors(input, '_originalChars');
1634
+ return preview.own &&
1635
+ !preview.accessor &&
1636
+ typeof preview.value === 'string' &&
1637
+ originalChars.own &&
1638
+ !originalChars.accessor &&
1639
+ typeof originalChars.value === 'number' &&
1640
+ Number.isFinite(originalChars.value) &&
1641
+ originalChars.value >= 0
1642
+ ? { preview: preview.value, originalChars: originalChars.value }
1643
+ : undefined;
1644
+ }
1645
+
1507
1646
  function projectToolInputWithinLimit(
1508
1647
  input: unknown,
1509
1648
  maxChars: number
1510
1649
  ): ToolInputProjection {
1511
1650
  const normalizedMaxChars = normalizeToolInputLimit(maxChars);
1512
- const serialized = serializeStructuredValueBounded(input, normalizedMaxChars);
1651
+ const priorTruncation = readBoundedTruncationValue(input);
1652
+ if (priorTruncation != null) {
1653
+ const serializedLength = serializeStructuredValueBounded(
1654
+ input,
1655
+ normalizedMaxChars
1656
+ );
1657
+ if (!serializedLength.truncated) {
1658
+ return { value: input, changed: false };
1659
+ }
1660
+ return {
1661
+ value: createBoundedTruncationValue(
1662
+ priorTruncation.preview,
1663
+ priorTruncation.originalChars,
1664
+ normalizedMaxChars
1665
+ ),
1666
+ changed: true,
1667
+ };
1668
+ }
1669
+ const serialized = serializeStructuredValueBounded(
1670
+ input,
1671
+ normalizedMaxChars,
1672
+ normalizedMaxChars
1673
+ );
1513
1674
  if (serialized.truncated) {
1514
1675
  return {
1515
1676
  value: createBoundedTruncationValue(
1516
- serialized.content,
1677
+ serialized.prefix,
1517
1678
  serialized.originalChars,
1518
1679
  normalizedMaxChars
1519
1680
  ),
@@ -1550,13 +1711,14 @@ export function serializeToolCallInput(
1550
1711
  const projected = projectToolInputWithinLimit(input, normalizedMaxChars);
1551
1712
  const serialized = serializeStructuredValueBounded(
1552
1713
  projected.value,
1714
+ normalizedMaxChars,
1553
1715
  normalizedMaxChars
1554
1716
  );
1555
1717
  if (!serialized.truncated) {
1556
1718
  return serialized.content === 'undefined' ? 'null' : serialized.content;
1557
1719
  }
1558
1720
  const fallback = createBoundedTruncationValue(
1559
- serialized.content,
1721
+ serialized.prefix,
1560
1722
  serialized.originalChars,
1561
1723
  normalizedMaxChars
1562
1724
  );
@@ -1616,10 +1778,9 @@ function projectInlineToolInput(
1616
1778
  nestedChanges
1617
1779
  );
1618
1780
  if (Object.keys(nestedChanges).length > 0) {
1619
- changes.tool_call = cloneWithProjectedProperties(
1620
- nestedProperty.value,
1621
- nestedChanges
1622
- );
1781
+ changes.tool_call = isProxy(nestedProperty.value)
1782
+ ? { args: ACCESSOR_INPUT_PLACEHOLDER }
1783
+ : cloneWithProjectedProperties(nestedProperty.value, nestedChanges);
1623
1784
  } else if (!nestedProperty.own) {
1624
1785
  changes.tool_call = nestedProperty.value;
1625
1786
  }
@@ -1640,12 +1801,58 @@ function projectSerializedArguments(
1640
1801
  if (typeof value === 'string' && value.length <= normalizedMaxChars) {
1641
1802
  return { value, changed: false };
1642
1803
  }
1804
+ if (
1805
+ typeof value === 'string' &&
1806
+ value.includes('"_truncated"') &&
1807
+ value.includes('"_originalChars"')
1808
+ ) {
1809
+ try {
1810
+ const priorTruncation = readBoundedTruncationValue(JSON.parse(value));
1811
+ if (priorTruncation != null) {
1812
+ return {
1813
+ value: JSON.stringify(
1814
+ createBoundedTruncationValue(
1815
+ priorTruncation.preview,
1816
+ priorTruncation.originalChars,
1817
+ normalizedMaxChars
1818
+ )
1819
+ ),
1820
+ changed: true,
1821
+ };
1822
+ }
1823
+ } catch {
1824
+ // Fall through to the accessor-safe serializer for malformed JSON.
1825
+ }
1826
+ }
1643
1827
  return {
1644
1828
  value: serializeToolCallInput(value, normalizedMaxChars),
1645
1829
  changed: true,
1646
1830
  };
1647
1831
  }
1648
1832
 
1833
+ const TRUNCATED_STRING_INPUT_PATTERN = /\n… \[truncated: (\d+) chars\]$/u;
1834
+
1835
+ function projectStringInputWithinLimit(
1836
+ value: string,
1837
+ maxChars: number
1838
+ ): { value: string; changed: boolean } {
1839
+ const normalizedMaxChars = normalizeToolInputLimit(maxChars);
1840
+ if (value.length <= normalizedMaxChars) {
1841
+ return { value, changed: false };
1842
+ }
1843
+ const match = TRUNCATED_STRING_INPUT_PATTERN.exec(value);
1844
+ const originalChars = match == null ? value.length : Number(match[1]);
1845
+ const prefix = match == null ? value : value.slice(0, match.index);
1846
+ const marker = `\n… [truncated: ${originalChars} chars]`;
1847
+ return {
1848
+ value:
1849
+ marker.length >= normalizedMaxChars
1850
+ ? prefix.slice(0, normalizedMaxChars)
1851
+ : prefix.slice(0, normalizedMaxChars - marker.length) + marker,
1852
+ changed: true,
1853
+ };
1854
+ }
1855
+
1649
1856
  function selectProjectedSerializedArguments(
1650
1857
  property: PropertyRead,
1651
1858
  canonical: string | undefined,
@@ -1862,10 +2069,7 @@ function projectResponsesOutput(
1862
2069
  } else if (type === 'custom_tool_call') {
1863
2070
  const value =
1864
2071
  typeof source === 'string'
1865
- ? truncateToolResultContent(
1866
- source,
1867
- normalizeToolInputLimit(maxChars)
1868
- )
2072
+ ? projectStringInputWithinLimit(source, maxChars).value
1869
2073
  : serializeToolCallInput(source, maxChars);
1870
2074
  projectedInput = {
1871
2075
  value,
@@ -1896,35 +2100,15 @@ function projectResponsesOutput(
1896
2100
  };
1897
2101
  }
1898
2102
 
1899
- /** Per-input cap: 15% of context at ~4 chars/token, never above 200K chars. */
1900
- export function calculateMaxToolCallInputChars(
1901
- maxContextTokens?: number
1902
- ): number {
1903
- if (maxContextTokens == null || maxContextTokens <= 0) {
1904
- return HARD_MAX_TOOL_CALL_INPUT_CHARS;
1905
- }
1906
- return Math.max(
1907
- MIN_JSON_VALUE_CHARS,
1908
- Math.min(
1909
- Math.floor(maxContextTokens * 0.15) * 4,
1910
- HARD_MAX_TOOL_CALL_INPUT_CHARS
1911
- )
1912
- );
1913
- }
1914
-
1915
- /**
1916
- * Projects historical tool-call inputs into a provider-safe bounded form.
1917
- * Returns the original array when no message changes and otherwise clones only
1918
- * the array and AI messages whose inline input or `tool_calls` args changed.
1919
- */
1920
2103
  function projectToolCallInputsInternal(
1921
2104
  messages: BaseMessage[],
1922
2105
  maxInputChars: number,
1923
- dropIncompleteStreamContent: boolean
2106
+ dropIncompleteStreamContent: boolean,
2107
+ fromIndex = 0
1924
2108
  ): BaseMessage[] {
1925
2109
  const normalizedMaxInputChars = normalizeToolInputLimit(maxInputChars);
1926
2110
  let projectedMessages: BaseMessage[] | undefined;
1927
- for (let i = 0; i < messages.length; i++) {
2111
+ for (let i = fromIndex; i < messages.length; i++) {
1928
2112
  const message = messages[i];
1929
2113
  const messageRole = (message as BaseMessage & { role?: unknown }).role;
1930
2114
  if (message.getType() !== 'ai' && messageRole !== 'assistant') {
@@ -2091,9 +2275,15 @@ function projectToolCallInputsInternal(
2091
2275
  /** Projects all historical tool-call input representations to bounded values. */
2092
2276
  export function projectToolCallInputs(
2093
2277
  messages: BaseMessage[],
2094
- maxInputChars: number
2278
+ maxInputChars: number,
2279
+ fromIndex = 0
2095
2280
  ): BaseMessage[] {
2096
- return projectToolCallInputsInternal(messages, maxInputChars, false);
2281
+ return projectToolCallInputsInternal(
2282
+ messages,
2283
+ maxInputChars,
2284
+ false,
2285
+ fromIndex
2286
+ );
2097
2287
  }
2098
2288
 
2099
2289
  /**
@@ -2107,32 +2297,84 @@ export function projectToolMessagesForProvider(
2107
2297
  return projectToolCallInputsInternal(messages, maxInputChars, true);
2108
2298
  }
2109
2299
 
2110
- export function preFlightTruncateToolCallInputs(params: {
2300
+ function applyToolCallInputCaps(params: {
2301
+ canonicalMessages?: BaseMessage[];
2111
2302
  messages: BaseMessage[];
2112
- maxContextTokens: number;
2303
+ maxInputChars: number;
2113
2304
  indexTokenCountMap: Record<string, number | undefined>;
2114
2305
  tokenCounter: TokenCounter;
2306
+ fromIndex?: number;
2115
2307
  }): number {
2116
- const { messages, maxContextTokens, indexTokenCountMap, tokenCounter } =
2117
- params;
2118
- const maxInputChars = calculateMaxToolCallInputChars(maxContextTokens);
2119
- const projected = projectToolCallInputs(messages, maxInputChars);
2120
- if (projected === messages) {
2308
+ const { messages, maxInputChars, indexTokenCountMap, tokenCounter } = params;
2309
+ const fromIndex = params.fromIndex ?? 0;
2310
+ const sourceMessages = params.canonicalMessages ?? messages;
2311
+ const projected = projectToolCallInputs(
2312
+ sourceMessages,
2313
+ maxInputChars,
2314
+ fromIndex
2315
+ );
2316
+ if (projected === sourceMessages) {
2121
2317
  return 0;
2122
2318
  }
2123
2319
 
2124
2320
  let truncatedCount = 0;
2125
- for (let i = 0; i < messages.length; i++) {
2321
+ for (let i = fromIndex; i < messages.length; i++) {
2322
+ if (projected[i] === sourceMessages[i]) {
2323
+ continue;
2324
+ }
2126
2325
  if (projected[i] === messages[i]) {
2127
2326
  continue;
2128
2327
  }
2129
- messages[i] = projected[i];
2130
- indexTokenCountMap[i] = tokenCounter(projected[i]);
2328
+ const current = messages[i] as AIMessage;
2329
+ const canonical = sourceMessages[i] as AIMessage;
2330
+ const capped = projected[i] as AIMessage;
2331
+ const changes: Record<string, unknown> = {};
2332
+ if (capped.content !== canonical.content) {
2333
+ changes.content = capped.content;
2334
+ }
2335
+ if (capped.tool_calls !== canonical.tool_calls) {
2336
+ changes.tool_calls = capped.tool_calls;
2337
+ }
2338
+ const additionalKwargsChanges: Record<string, unknown> = {};
2339
+ for (const key of ['tool_calls', 'function_call']) {
2340
+ if (capped.additional_kwargs[key] !== canonical.additional_kwargs[key]) {
2341
+ additionalKwargsChanges[key] = capped.additional_kwargs[key];
2342
+ }
2343
+ }
2344
+ if (Object.keys(additionalKwargsChanges).length > 0) {
2345
+ changes.additional_kwargs = cloneWithProjectedProperties(
2346
+ current.additional_kwargs,
2347
+ additionalKwargsChanges
2348
+ );
2349
+ }
2350
+ if (capped.response_metadata.output !== canonical.response_metadata.output) {
2351
+ changes.response_metadata = cloneWithProjectedProperties(
2352
+ current.response_metadata,
2353
+ { output: capped.response_metadata.output }
2354
+ );
2355
+ }
2356
+ const merged = cloneWithProjectedProperties(current, changes);
2357
+ messages[i] = merged;
2358
+ indexTokenCountMap[i] = tokenCounter(merged);
2131
2359
  truncatedCount++;
2132
2360
  }
2133
2361
  return truncatedCount;
2134
2362
  }
2135
2363
 
2364
+ export function preFlightTruncateToolCallInputs(params: {
2365
+ messages: BaseMessage[];
2366
+ maxContextTokens: number;
2367
+ indexTokenCountMap: Record<string, number | undefined>;
2368
+ tokenCounter: TokenCounter;
2369
+ }): number {
2370
+ return applyToolCallInputCaps({
2371
+ messages: params.messages,
2372
+ maxInputChars: calculateMaxToolCallInputChars(params.maxContextTokens),
2373
+ indexTokenCountMap: params.indexTokenCountMap,
2374
+ tokenCounter: params.tokenCounter,
2375
+ });
2376
+ }
2377
+
2136
2378
  type ThinkingBlocks = {
2137
2379
  thinking_blocks?: Array<{
2138
2380
  type: 'thinking';
@@ -2175,6 +2417,7 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2175
2417
  * Self-seeds from provider observations within the run. */
2176
2418
  let bestInstructionOverhead: number | undefined;
2177
2419
  const reconciledLegacyAiMessages = new WeakSet<BaseMessage>();
2420
+ const canonicalizedToolCallMessages = new WeakSet<BaseMessage>();
2178
2421
  let bestVarianceAbs = Infinity;
2179
2422
  /** Local estimate at the time bestInstructionOverhead was observed.
2180
2423
  * Used to invalidate the cached overhead when instructions change
@@ -2186,6 +2429,23 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2186
2429
  * pruner is recreated after summarization. */
2187
2430
  const originalToolContent = new Map<number, string>();
2188
2431
  let originalToolContentSize = 0;
2432
+ /** Recovers canonical sources for direct callers that reuse this mutating
2433
+ * projection without passing graph-owned history explicitly. */
2434
+ const canonicalByProjection = new WeakMap<BaseMessage, BaseMessage>();
2435
+ /** Latched fading tier; caps derive from it alone so bytes stay stable. */
2436
+ let fadingTier = seedFadingTier(
2437
+ factoryParams.maxTokens,
2438
+ factoryParams.fadingTier
2439
+ );
2440
+ let restoredTierPending = isFadingTier(factoryParams.fadingTier);
2441
+ /** Widest exchange seen so far; updated only from the appended suffix. */
2442
+ let maxToolExchangeWidth = 1;
2443
+ let toolExchangeWidthThrough = 0;
2444
+ let toolExchangeWidthSources: BaseMessage[] = [];
2445
+ /** Fresh results and inputs below this index already carry the tier's caps. */
2446
+ let fadedThrough = 0;
2447
+ /** Consumed results below this index already carry the tier's masked cap. */
2448
+ let maskedThrough = 0;
2189
2449
  const contextPruningSettings = resolveContextPruningSettings(
2190
2450
  factoryParams.contextPruningConfig
2191
2451
  );
@@ -2200,12 +2460,62 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2200
2460
  originalToolContent?: Map<number, string>;
2201
2461
  newOriginalToolContent?: Map<number, string>;
2202
2462
  calibrationRatio?: number;
2463
+ /** Latched fading tier after this call; hosts persist it beside calibrationRatio. */
2464
+ fadingTier: FadingTier;
2203
2465
  resolvedInstructionOverhead?: number;
2204
2466
  /** Usable budget this call: maxTokens minus output reserve */
2205
2467
  contextBudget?: number;
2206
2468
  /** Calibrated instruction overhead actually applied this call */
2207
2469
  effectiveInstructionTokens?: number;
2208
2470
  } {
2471
+ const suppliedCanonicalMessages = params.canonicalMessages;
2472
+ const derivesCanonicalHistory = suppliedCanonicalMessages == null;
2473
+ const canonicalMessages =
2474
+ suppliedCanonicalMessages ??
2475
+ params.messages.map((message) => {
2476
+ const recovered = canonicalByProjection.get(message);
2477
+ if (recovered != null) {
2478
+ return recovered;
2479
+ }
2480
+ return Array.isArray(message.content)
2481
+ ? cloneMessage(message, [...message.content])
2482
+ : message;
2483
+ });
2484
+ const priorWidthLength = toolExchangeWidthSources.length;
2485
+ const widthPrefixChanged = derivesCanonicalHistory
2486
+ ? canonicalMessages.length < priorWidthLength ||
2487
+ toolExchangeWidthSources.some(
2488
+ (source, index) => canonicalMessages[index] !== source
2489
+ )
2490
+ : canonicalMessages.length < priorWidthLength ||
2491
+ (params.canonicalPrefixStable === true
2492
+ ? priorWidthLength > 0 &&
2493
+ canonicalMessages[priorWidthLength - 1] !==
2494
+ toolExchangeWidthSources[priorWidthLength - 1]
2495
+ : toolExchangeWidthSources.some(
2496
+ (source, index) => canonicalMessages[index] !== source
2497
+ ));
2498
+ if (widthPrefixChanged) {
2499
+ maxToolExchangeWidth = 1;
2500
+ toolExchangeWidthThrough = 0;
2501
+ toolExchangeWidthSources = [];
2502
+ fadedThrough = 0;
2503
+ maskedThrough = 0;
2504
+ originalToolContent.clear();
2505
+ originalToolContentSize = 0;
2506
+ }
2507
+ for (
2508
+ let i = toolExchangeWidthThrough;
2509
+ i < canonicalMessages.length;
2510
+ i++
2511
+ ) {
2512
+ maxToolExchangeWidth = Math.max(
2513
+ maxToolExchangeWidth,
2514
+ getToolCallIds(canonicalMessages[i]).size
2515
+ );
2516
+ toolExchangeWidthSources[i] = canonicalMessages[i];
2517
+ }
2518
+ toolExchangeWidthThrough = canonicalMessages.length;
2209
2519
  let newOriginalToolContent: Map<number, string> | undefined;
2210
2520
  if (params.messages.length === 0) {
2211
2521
  /** Post-compaction calls still invoke the model — report the same
@@ -2232,6 +2542,7 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2232
2542
  emptyBudget - emptyInstructionTokens - emptyReplyPrimerTokens
2233
2543
  ),
2234
2544
  calibrationRatio,
2545
+ fadingTier,
2235
2546
  resolvedInstructionOverhead: bestInstructionOverhead,
2236
2547
  contextBudget: emptyBudget,
2237
2548
  effectiveInstructionTokens: emptyInstructionTokens,
@@ -2326,10 +2637,28 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2326
2637
  };
2327
2638
  for (let i = 0; i < params.messages.length; i++) {
2328
2639
  let message = params.messages[i];
2329
- const cachedCount = indexTokenCountMap[i];
2640
+ let cachedCount = indexTokenCountMap[i];
2330
2641
  const messageType = message.getType();
2331
2642
  const messageRole = (message as BaseMessage & { role?: unknown }).role;
2332
2643
  const isAssistant = messageType === 'ai' || messageRole === 'assistant';
2644
+ if (isAssistant && !canonicalizedToolCallMessages.has(message)) {
2645
+ const [canonicalized] = projectToolCallInputs(
2646
+ [message],
2647
+ HARD_MAX_TOOL_CALL_INPUT_CHARS
2648
+ );
2649
+ if (canonicalized !== message) {
2650
+ message = canonicalized;
2651
+ params.messages[i] = message;
2652
+ const canonicalizedCount = factoryParams.tokenCounter(message);
2653
+ indexTokenCountMap[i] = canonicalizedCount;
2654
+ totalTokens += canonicalizedCount - (cachedCount ?? 0);
2655
+ cachedCount = canonicalizedCount;
2656
+ if (i >= lastTurnStartIndex) {
2657
+ newOutputs.add(i);
2658
+ }
2659
+ }
2660
+ canonicalizedToolCallMessages.add(message);
2661
+ }
2333
2662
  const legacyFunctionCall = isAssistant
2334
2663
  ? readPropertyWithoutAccessors(
2335
2664
  message.additional_kwargs,
@@ -2347,6 +2676,13 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2347
2676
  if (projected !== message) {
2348
2677
  message = projected;
2349
2678
  params.messages[i] = message;
2679
+ const projectedCount = factoryParams.tokenCounter(message);
2680
+ indexTokenCountMap[i] = projectedCount;
2681
+ totalTokens += projectedCount - (cachedCount ?? 0);
2682
+ cachedCount = projectedCount;
2683
+ if (i >= lastTurnStartIndex) {
2684
+ newOutputs.add(i);
2685
+ }
2350
2686
  }
2351
2687
  reconciledLegacyAiMessages.add(message);
2352
2688
  if (cachedCount !== undefined || i < lastTurnStartIndex) {
@@ -2524,16 +2860,6 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2524
2860
 
2525
2861
  let calibratedTotalTokens = Math.round(totalTokens * calibrationRatio);
2526
2862
 
2527
- factoryParams.log?.('debug', 'Budget', {
2528
- maxTokens: factoryParams.maxTokens,
2529
- pruningBudget,
2530
- effectiveMax: effectiveMaxTokens,
2531
- instructionTokens: currentInstructionTokens,
2532
- messageCount: params.messages.length,
2533
- calibratedTotalTokens,
2534
- calibrationRatio: Math.round(calibrationRatio * 100) / 100,
2535
- });
2536
-
2537
2863
  // When instructions alone consume the entire budget, no message can
2538
2864
  // fit regardless of truncation. Short-circuit: yield all messages for
2539
2865
  // summarization and return an empty context so the Graph can route to
@@ -2564,6 +2890,7 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2564
2890
  contextPressure:
2565
2891
  pruningBudget > 0 ? calibratedTotalTokens / pruningBudget : 0,
2566
2892
  calibrationRatio,
2893
+ fadingTier,
2567
2894
  resolvedInstructionOverhead: bestInstructionOverhead,
2568
2895
  contextBudget: pruningBudget,
2569
2896
  effectiveInstructionTokens: currentInstructionTokens,
@@ -2572,25 +2899,14 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2572
2899
 
2573
2900
  // ---------------------------------------------------------------------------
2574
2901
  // Progressive context fading — inspired by Claude Code's staged compaction.
2575
- // Below 80%: no modifications, tool results retain full size.
2576
- // Above 80%: graduated truncation with increasing aggression per pressure band.
2577
- // Recency weighting ensures older results fade first, newer results last.
2578
- //
2579
- // At the gentlest level, truncation preserves most content (head+tail).
2580
- // At the most aggressive level, the result is effectively a one-line placeholder.
2581
- //
2582
- // 80%: gentle — budget factor 1.0, oldest get light truncation
2583
- // 85%: moderate — budget factor 0.50, older results shrink significantly
2584
- // 90%: aggressive — budget factor 0.20, most results heavily truncated
2585
- // 99%: emergency — budget factor 0.05, effectively placeholders for old results
2902
+ // Every cap comes from the latched fading tier (see ./fading.ts): the fit
2903
+ // rung keeps a single tool result within the effective budget, pressure
2904
+ // bands deepen the rung when summarization is off, and masking (80%+)
2905
+ // shrinks consumed results the model has already answered. The tier only
2906
+ // ever deepens, so a historical tool result maps to the same bytes on
2907
+ // every call and provider prompt-cache prefixes survive from turn to turn;
2908
+ // only escalation rewrites them.
2586
2909
  // ---------------------------------------------------------------------------
2587
- totalTokens = sumTokenCounts(indexTokenCountMap, params.messages.length);
2588
- calibratedTotalTokens = Math.round(totalTokens * calibrationRatio);
2589
- const contextPressure =
2590
- pruningBudget > 0 ? calibratedTotalTokens / pruningBudget : 0;
2591
- let preFlightResultCount = 0;
2592
- let preFlightInputCount = 0;
2593
-
2594
2910
  // -----------------------------------------------------------------------
2595
2911
  // Observation masking (80%+ pressure, both paths):
2596
2912
  // Replace consumed ToolMessage content with tight head+tail placeholders.
@@ -2601,138 +2917,141 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2601
2917
  // When summarization is enabled, snapshot messages first so the
2602
2918
  // summarizer can see the full originals when compaction fires.
2603
2919
  // -----------------------------------------------------------------------
2604
- let observationsMasked = 0;
2920
+ const storeOriginalToolContent = (index: number, content: string): void => {
2921
+ originalToolContentSize += content.length;
2922
+ if (newOriginalToolContent == null) {
2923
+ newOriginalToolContent = new Map();
2924
+ }
2925
+ newOriginalToolContent.set(index, content);
2926
+ while (
2927
+ originalToolContentSize > ORIGINAL_CONTENT_MAX_CHARS &&
2928
+ originalToolContent.size > 0
2929
+ ) {
2930
+ const oldest = originalToolContent.keys().next();
2931
+ if (oldest.done === true) {
2932
+ break;
2933
+ }
2934
+ const removed = originalToolContent.get(oldest.value);
2935
+ if (removed != null) {
2936
+ originalToolContentSize -= removed.length;
2937
+ }
2938
+ originalToolContent.delete(oldest.value);
2939
+ }
2940
+ };
2605
2941
 
2606
- if (contextPressure >= PRESSURE_THRESHOLD_MASKING) {
2607
- const rawMessageBudget =
2608
- calibrationRatio > 0
2609
- ? Math.floor(effectiveMaxTokens / calibrationRatio)
2610
- : effectiveMaxTokens;
2611
- // When summarization is enabled, use half the reserve ratio as extra
2612
- // masking headroom — the LLM keeps more context while the summarizer
2613
- // gets full content from originalToolContent regardless. The remaining
2614
- // half of the reserve covers estimation errors.
2615
- const reserveHeadroom =
2616
- factoryParams.summarizationEnabled === true
2617
- ? Math.floor(
2618
- rawMessageBudget *
2619
- (factoryParams.reserveRatio ?? DEFAULT_RESERVE_RATIO) *
2620
- 0.5
2621
- )
2622
- : 0;
2623
- observationsMasked = maskConsumedToolResults({
2942
+ let restoredFading: FadingApplyResult = {
2943
+ truncated: 0,
2944
+ inputs: 0,
2945
+ masked: 0,
2946
+ consumedBoundary: 0,
2947
+ };
2948
+ if (restoredTierPending) {
2949
+ restoredFading = applyFadingCaps({
2624
2950
  messages: params.messages,
2951
+ canonicalMessages,
2625
2952
  indexTokenCountMap,
2626
2953
  tokenCounter: factoryParams.tokenCounter,
2627
- availableRawBudget: rawMessageBudget + reserveHeadroom,
2954
+ caps: resolveFadingCaps(fadingTier, factoryParams.maxToolResultChars),
2955
+ masked: fadingTier.masked,
2628
2956
  originalContentStore:
2629
2957
  factoryParams.summarizationEnabled === true
2630
2958
  ? originalToolContent
2631
2959
  : undefined,
2632
2960
  onContentStored:
2633
2961
  factoryParams.summarizationEnabled === true
2634
- ? (index: number, content: string): void => {
2635
- originalToolContentSize += content.length;
2636
- if (newOriginalToolContent == null) {
2637
- newOriginalToolContent = new Map();
2638
- }
2639
- newOriginalToolContent.set(index, content);
2640
- while (
2641
- originalToolContentSize > ORIGINAL_CONTENT_MAX_CHARS &&
2642
- originalToolContent.size > 0
2643
- ) {
2644
- const oldest = originalToolContent.keys().next();
2645
- if (oldest.done === true) {
2646
- break;
2647
- }
2648
- const removed = originalToolContent.get(oldest.value);
2649
- if (removed != null) {
2650
- originalToolContentSize -= removed.length;
2651
- }
2652
- originalToolContent.delete(oldest.value);
2653
- }
2654
- }
2962
+ ? storeOriginalToolContent
2655
2963
  : undefined,
2656
2964
  });
2657
- if (observationsMasked > 0) {
2658
- cumulativeRawSent = 0;
2659
- cumulativeProviderReported = 0;
2660
- }
2965
+ fadedThrough = params.messages.length;
2966
+ maskedThrough = restoredFading.consumedBoundary;
2967
+ restoredTierPending = false;
2661
2968
  }
2662
2969
 
2663
- if (
2664
- contextPressure >= PRESSURE_THRESHOLD_MASKING &&
2665
- factoryParams.summarizationEnabled !== true
2666
- ) {
2667
- const budgetFactor =
2668
- PRESSURE_BANDS.find(
2669
- ([threshold]) => contextPressure >= threshold
2670
- )?.[1] ?? 1.0;
2671
-
2672
- const baseBudget = Math.max(
2673
- 1024,
2674
- Math.floor(effectiveMaxTokens * budgetFactor)
2675
- );
2970
+ totalTokens = sumTokenCounts(indexTokenCountMap, params.messages.length);
2971
+ calibratedTotalTokens = Math.round(totalTokens * calibrationRatio);
2972
+ const contextPressure =
2973
+ pruningBudget > 0 ? calibratedTotalTokens / pruningBudget : 0;
2676
2974
 
2677
- preFlightResultCount = preFlightTruncateToolResults({
2678
- messages: params.messages,
2679
- maxContextTokens: baseBudget,
2680
- indexTokenCountMap,
2681
- tokenCounter: factoryParams.tokenCounter,
2682
- });
2975
+ factoryParams.log?.('debug', 'Budget', {
2976
+ maxTokens: factoryParams.maxTokens,
2977
+ pruningBudget,
2978
+ effectiveMax: effectiveMaxTokens,
2979
+ instructionTokens: currentInstructionTokens,
2980
+ messageCount: params.messages.length,
2981
+ calibratedTotalTokens,
2982
+ calibrationRatio: Math.round(calibrationRatio * 100) / 100,
2983
+ });
2683
2984
 
2684
- preFlightInputCount = preFlightTruncateToolCallInputs({
2985
+ /** Advances the tier from the signals and applies its caps; escalation rescans everything. */
2986
+ const fade = (signals: FadingSignals): FadingApplyResult => {
2987
+ const nextTier = resolveFadingTier(
2988
+ fadingTier,
2989
+ factoryParams.maxTokens,
2990
+ signals,
2991
+ factoryParams.maxToolResultChars
2992
+ );
2993
+ if (nextTier !== fadingTier) {
2994
+ fadingTier = nextTier;
2995
+ fadedThrough = 0;
2996
+ maskedThrough = 0;
2997
+ }
2998
+ const result = applyFadingCaps({
2685
2999
  messages: params.messages,
2686
- maxContextTokens: baseBudget,
3000
+ canonicalMessages,
2687
3001
  indexTokenCountMap,
2688
3002
  tokenCounter: factoryParams.tokenCounter,
3003
+ caps: resolveFadingCaps(fadingTier, factoryParams.maxToolResultChars),
3004
+ masked: fadingTier.masked,
3005
+ fromIndex: fadedThrough,
3006
+ maskedFromIndex: maskedThrough,
3007
+ originalContentStore:
3008
+ factoryParams.summarizationEnabled === true
3009
+ ? originalToolContent
3010
+ : undefined,
3011
+ onContentStored:
3012
+ factoryParams.summarizationEnabled === true
3013
+ ? storeOriginalToolContent
3014
+ : undefined,
2689
3015
  });
3016
+ fadedThrough = params.messages.length;
3017
+ maskedThrough = result.consumedBoundary;
3018
+ return result;
3019
+ };
3020
+
3021
+ const fadingSignals: FadingSignals = {
3022
+ contextPressure,
3023
+ effectiveRawTokens:
3024
+ calibrationRatio > 0
3025
+ ? Math.floor(effectiveMaxTokens / calibrationRatio)
3026
+ : effectiveMaxTokens,
3027
+ summarizationEnabled: factoryParams.summarizationEnabled === true,
3028
+ toolExchangeWidth: maxToolExchangeWidth,
3029
+ };
3030
+ const faded = fade(fadingSignals);
3031
+ const observationsMasked = restoredFading.masked + faded.masked;
3032
+ const preFlightResultCount = restoredFading.truncated + faded.truncated;
3033
+ const preFlightInputCount = restoredFading.inputs + faded.inputs;
3034
+ if (observationsMasked > 0) {
3035
+ cumulativeRawSent = 0;
3036
+ cumulativeProviderReported = 0;
2690
3037
  }
3038
+
2691
3039
  if (
2692
3040
  factoryParams.contextPruningConfig?.enabled === true &&
2693
3041
  factoryParams.summarizationEnabled !== true
2694
3042
  ) {
2695
3043
  applyContextPruning({
2696
3044
  messages: params.messages,
3045
+ canonicalMessages,
2697
3046
  indexTokenCountMap,
2698
3047
  tokenCounter: factoryParams.tokenCounter,
2699
3048
  resolvedSettings: contextPruningSettings,
2700
3049
  });
2701
3050
  }
2702
-
2703
- // Fit-to-budget: when summarization is enabled and individual messages
2704
- // exceed the effective budget, truncate them so every message can fit in
2705
- // a single context slot. Without this, oversized tool results (e.g.
2706
- // take_snapshot at 9K chars) cause empty context → emergency truncation
2707
- // → immediate re-summarization after just one tool call.
2708
- //
2709
- // This is NOT the lossy position-based fading above — it only targets
2710
- // messages that individually exceed the budget, using the full effective
2711
- // budget as the cap (not a pressure-scaled fraction).
2712
- // Fit-to-budget caps are in raw space (divide by ratio) so that after
2713
- // calibration the truncated results actually fit within the budget.
2714
- const rawSpaceEffectiveMax =
2715
- calibrationRatio > 0
2716
- ? Math.round(effectiveMaxTokens / calibrationRatio)
2717
- : effectiveMaxTokens;
2718
-
2719
- if (
2720
- factoryParams.summarizationEnabled === true &&
2721
- rawSpaceEffectiveMax > 0
2722
- ) {
2723
- preFlightResultCount = preFlightTruncateToolResults({
2724
- messages: params.messages,
2725
- maxContextTokens: rawSpaceEffectiveMax,
2726
- indexTokenCountMap,
2727
- tokenCounter: factoryParams.tokenCounter,
2728
- });
2729
-
2730
- preFlightInputCount = preFlightTruncateToolCallInputs({
2731
- messages: params.messages,
2732
- maxContextTokens: rawSpaceEffectiveMax,
2733
- indexTokenCountMap,
2734
- tokenCounter: factoryParams.tokenCounter,
2735
- });
3051
+ if (derivesCanonicalHistory) {
3052
+ for (let i = 0; i < params.messages.length; i++) {
3053
+ canonicalByProjection.set(params.messages[i], canonicalMessages[i]);
3054
+ }
2736
3055
  }
2737
3056
 
2738
3057
  const preTruncationTotalTokens = totalTokens;
@@ -2783,6 +3102,7 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2783
3102
  originalToolContent.size > 0 ? originalToolContent : undefined,
2784
3103
  newOriginalToolContent,
2785
3104
  calibrationRatio,
3105
+ fadingTier,
2786
3106
  resolvedInstructionOverhead: bestInstructionOverhead,
2787
3107
  contextBudget: pruningBudget,
2788
3108
  effectiveInstructionTokens: currentInstructionTokens,
@@ -2856,106 +3176,14 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2856
3176
  appendItems(messagesToRefine, droppedMessages);
2857
3177
  }
2858
3178
 
2859
- // ---------------------------------------------------------------
2860
- // Fallback fading: when summarization skipped fading earlier and
2861
- // pruning still produced an empty context, apply lossy pressure-band
2862
- // fading and retry. This is a last resort before emergency truncation
2863
- // — the summarizer already saw the full messages, so fading the
2864
- // surviving context for the LLM is acceptable.
2865
- // ---------------------------------------------------------------
2866
- if (
2867
- context.length === 0 &&
2868
- params.messages.length > 0 &&
2869
- effectiveMaxTokens > 0 &&
2870
- factoryParams.summarizationEnabled === true
2871
- ) {
2872
- const fadingBudget = Math.max(1024, effectiveMaxTokens);
2873
-
2874
- factoryParams.log?.(
2875
- 'debug',
2876
- 'Fallback fading — empty context with summarization',
2877
- {
2878
- messageCount: params.messages.length,
2879
- effectiveMaxTokens,
2880
- fadingBudget,
2881
- }
2882
- );
2883
-
2884
- const fadedMessages = [...params.messages];
2885
- const preFadingTokenCounts: Record<string, number | undefined> = {};
2886
- for (let i = 0; i < params.messages.length; i++) {
2887
- preFadingTokenCounts[i] = indexTokenCountMap[i];
2888
- }
2889
-
2890
- preFlightTruncateToolResults({
2891
- messages: fadedMessages,
2892
- maxContextTokens: fadingBudget,
2893
- indexTokenCountMap,
2894
- tokenCounter: factoryParams.tokenCounter,
2895
- });
2896
- preFlightTruncateToolCallInputs({
2897
- messages: fadedMessages,
2898
- maxContextTokens: fadingBudget,
2899
- indexTokenCountMap,
2900
- tokenCounter: factoryParams.tokenCounter,
2901
- });
2902
-
2903
- const fadingRetry = getMessagesWithinTokenLimit({
2904
- maxContextTokens: pruningBudget,
2905
- messages: fadedMessages,
2906
- indexTokenCountMap,
2907
- startType: params.startType,
2908
- thinkingEnabled: factoryParams.thinkingEnabled,
2909
- tokenCounter: factoryParams.tokenCounter,
2910
- instructionTokens: currentInstructionTokens,
2911
- reasoningType: usesBedrockThinking
2912
- ? ContentTypes.REASONING_CONTENT
2913
- : ContentTypes.THINKING,
2914
- thinkingStartIndex:
2915
- factoryParams.thinkingEnabled === true
2916
- ? runThinkingStartIndex
2917
- : undefined,
2918
- });
2919
-
2920
- const fadingRepaired = repairOrphanedToolMessages({
2921
- context: fadingRetry.context,
2922
- allMessages: fadedMessages,
2923
- tokenCounter: factoryParams.tokenCounter,
2924
- indexTokenCountMap,
2925
- });
2926
-
2927
- if (fadingRepaired.context.length > 0) {
2928
- context = fadingRepaired.context;
2929
- reclaimedTokens = fadingRepaired.reclaimedTokens;
2930
- appendItems(messagesToRefine, fadingRetry.messagesToRefine);
2931
- if (fadingRepaired.droppedMessages.length > 0) {
2932
- appendItems(messagesToRefine, fadingRepaired.droppedMessages);
2933
- }
2934
-
2935
- factoryParams.log?.('debug', 'Fallback fading recovered context', {
2936
- contextLength: context.length,
2937
- messagesToRefineCount: messagesToRefine.length,
2938
- remainingTokens: fadingRetry.remainingContextTokens,
2939
- });
2940
-
2941
- for (const [key, value] of Object.entries(preFadingTokenCounts)) {
2942
- indexTokenCountMap[key] = value;
2943
- }
2944
- } else {
2945
- for (const [key, value] of Object.entries(preFadingTokenCounts)) {
2946
- indexTokenCountMap[key] = value;
2947
- }
2948
- }
2949
- }
2950
-
2951
3179
  // ---------------------------------------------------------------
2952
3180
  // Emergency truncation: if pruning produced an empty context but
2953
- // messages exist, aggressively truncate all tool_call inputs and
2954
- // tool results, then retry. Budget is proportional to the
2955
- // effective token limit (~4 chars/token, spread across messages)
2956
- // with a floor of 200 chars so content is never completely blank.
2957
- // Uses head+tail so the model sees both what was called and the
2958
- // final outcome (e.g., return value at the end of a script eval).
3181
+ // messages exist, derive a deeper, temporary tier from a per-message
3182
+ // share of the effective budget (~4 chars/token, floor 200 chars),
3183
+ // apply it to a clone and retry. The latched tier is left alone: this
3184
+ // share depends on the message count, so latching it would pin every
3185
+ // future result to one transient event. The clone keeps graph state
3186
+ // intact for later turns where more budget may be available.
2959
3187
  // ---------------------------------------------------------------
2960
3188
  if (
2961
3189
  context.length === 0 &&
@@ -2966,6 +3194,23 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2966
3194
  effectiveMaxTokens / Math.max(1, params.messages.length)
2967
3195
  );
2968
3196
  const emergencyMaxChars = Math.max(200, perMessageTokenBudget * 4);
3197
+ const emergencyTier = resolveFadingTier(
3198
+ fadingTier,
3199
+ factoryParams.maxTokens,
3200
+ {
3201
+ ...fadingSignals,
3202
+ minRung: fadingRungForExchangeChars(
3203
+ factoryParams.maxTokens,
3204
+ emergencyMaxChars,
3205
+ factoryParams.maxToolResultChars
3206
+ ),
3207
+ },
3208
+ factoryParams.maxToolResultChars
3209
+ );
3210
+ const emergencyCaps = resolveFadingCaps(
3211
+ emergencyTier,
3212
+ factoryParams.maxToolResultChars
3213
+ );
2969
3214
 
2970
3215
  factoryParams.log?.(
2971
3216
  'warn',
@@ -2974,14 +3219,10 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2974
3219
  messageCount: params.messages.length,
2975
3220
  effectiveMax: effectiveMaxTokens,
2976
3221
  emergencyMaxChars,
3222
+ budgetTokens: emergencyCaps.budgetTokens,
2977
3223
  }
2978
3224
  );
2979
3225
 
2980
- // Clone the messages array so emergency truncation doesn't permanently
2981
- // mutate graph state. The originals remain intact for future turns
2982
- // where more budget may be available. Also snapshot indexTokenCountMap
2983
- // entries so the closure doesn't retain stale (too-small) counts for
2984
- // the original un-truncated messages on the next turn.
2985
3226
  const emergencyMessages = [...params.messages];
2986
3227
  const preEmergencyTokenCounts: Record<string, number | undefined> = {};
2987
3228
  for (let i = 0; i < params.messages.length; i++) {
@@ -2989,51 +3230,20 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
2989
3230
  }
2990
3231
 
2991
3232
  try {
2992
- let emergencyTruncatedCount = 0;
2993
- for (let i = 0; i < emergencyMessages.length; i++) {
2994
- const message = emergencyMessages[i];
2995
- if (
2996
- message.getType() === 'tool' &&
2997
- !isComputerCallOutputMessage(message)
2998
- ) {
2999
- const content = message.content;
3000
- if (getToolContentCharLength(content) > emergencyMaxChars) {
3001
- const compacted = compactToolContent(content, emergencyMaxChars);
3002
- if (!compacted.changed) {
3003
- continue;
3004
- }
3005
- const cloned = cloneToolMessageWithContent(
3006
- message as ToolMessage,
3007
- compacted.content
3008
- );
3009
- emergencyMessages[i] = cloned;
3010
- indexTokenCountMap[i] = factoryParams.tokenCounter(cloned);
3011
- emergencyTruncatedCount++;
3012
- }
3013
- }
3014
- }
3015
-
3016
- const projectedToolInputs = projectToolCallInputs(
3017
- emergencyMessages,
3018
- emergencyMaxChars
3019
- );
3020
- if (projectedToolInputs !== emergencyMessages) {
3021
- for (let i = 0; i < emergencyMessages.length; i++) {
3022
- if (projectedToolInputs[i] === emergencyMessages[i]) {
3023
- continue;
3024
- }
3025
- emergencyMessages[i] = projectedToolInputs[i];
3026
- indexTokenCountMap[i] = factoryParams.tokenCounter(
3027
- projectedToolInputs[i]
3028
- );
3029
- emergencyTruncatedCount++;
3030
- }
3031
- }
3233
+ const emergency = applyFadingCaps({
3234
+ messages: emergencyMessages,
3235
+ canonicalMessages,
3236
+ indexTokenCountMap,
3237
+ tokenCounter: factoryParams.tokenCounter,
3238
+ caps: emergencyCaps,
3239
+ masked: emergencyTier.masked,
3240
+ });
3032
3241
 
3033
3242
  factoryParams.log?.('info', 'Emergency truncation complete');
3034
3243
  factoryParams.log?.('debug', 'Emergency truncation details', {
3035
- truncatedCount: emergencyTruncatedCount,
3036
- emergencyMaxChars,
3244
+ truncatedCount:
3245
+ emergency.truncated + emergency.inputs + emergency.masked,
3246
+ budgetTokens: emergencyCaps.budgetTokens,
3037
3247
  });
3038
3248
 
3039
3249
  const retryResult = getMessagesWithinTokenLimit({
@@ -3062,6 +3272,9 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
3062
3272
 
3063
3273
  context = repaired.context;
3064
3274
  reclaimedTokens = repaired.reclaimedTokens;
3275
+ /** The retry supersedes the failed pass: messages now in context
3276
+ * must not also be handed to the summarizer. */
3277
+ messagesToRefine.length = 0;
3065
3278
  appendItems(messagesToRefine, retryResult.messagesToRefine);
3066
3279
  if (repaired.droppedMessages.length > 0) {
3067
3280
  appendItems(messagesToRefine, repaired.droppedMessages);
@@ -3075,8 +3288,6 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
3075
3288
  } finally {
3076
3289
  // Restore the closure's indexTokenCountMap to pre-emergency values so the
3077
3290
  // next turn counts old messages at their original (un-truncated) size.
3078
- // The emergency-truncated counts were only needed for this turn's
3079
- // getMessagesWithinTokenLimit retry.
3080
3291
  for (const [key, value] of Object.entries(preEmergencyTokenCounts)) {
3081
3292
  indexTokenCountMap[key] = value;
3082
3293
  }
@@ -3118,6 +3329,7 @@ export function createPruneMessages(factoryParams: PruneMessagesFactoryParams) {
3118
3329
  originalToolContent.size > 0 ? originalToolContent : undefined,
3119
3330
  newOriginalToolContent,
3120
3331
  calibrationRatio,
3332
+ fadingTier,
3121
3333
  resolvedInstructionOverhead: bestInstructionOverhead,
3122
3334
  contextBudget: pruningBudget,
3123
3335
  effectiveInstructionTokens: currentInstructionTokens,