@librechat/agents 3.8.0 → 3.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (248) hide show
  1. package/dist/cjs/agents/AgentContext.cjs +13 -3
  2. package/dist/cjs/agents/AgentContext.cjs.map +1 -1
  3. package/dist/cjs/graphs/Graph.cjs +176 -55
  4. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  5. package/dist/cjs/graphs/MultiAgentGraph.cjs +16 -6
  6. package/dist/cjs/graphs/MultiAgentGraph.cjs.map +1 -1
  7. package/dist/cjs/hitl/approvalReview.cjs +89 -0
  8. package/dist/cjs/hitl/approvalReview.cjs.map +1 -0
  9. package/dist/cjs/langfuse.cjs +42 -0
  10. package/dist/cjs/langfuse.cjs.map +1 -1
  11. package/dist/cjs/langfuseToolOutputTracing.cjs +42 -0
  12. package/dist/cjs/langfuseToolOutputTracing.cjs.map +1 -1
  13. package/dist/cjs/langfuseTraceShaping.cjs +7 -1
  14. package/dist/cjs/langfuseTraceShaping.cjs.map +1 -1
  15. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs +6 -10
  16. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs.map +1 -1
  17. package/dist/cjs/llm/bedrock/utils/message_inputs.cjs +2 -7
  18. package/dist/cjs/llm/bedrock/utils/message_inputs.cjs.map +1 -1
  19. package/dist/cjs/llm/contextPressureMeter.cjs +51 -10
  20. package/dist/cjs/llm/contextPressureMeter.cjs.map +1 -1
  21. package/dist/cjs/llm/init.cjs +1 -1
  22. package/dist/cjs/llm/invoke.cjs +2 -2
  23. package/dist/cjs/llm/openai/index.cjs +4 -3
  24. package/dist/cjs/llm/openai/index.cjs.map +1 -1
  25. package/dist/cjs/llm/openai/utils/index.cjs +5 -4
  26. package/dist/cjs/llm/openai/utils/index.cjs.map +1 -1
  27. package/dist/cjs/llm/preempt.cjs +3 -2
  28. package/dist/cjs/llm/preempt.cjs.map +1 -1
  29. package/dist/cjs/llm/prepareProviderRequest.cjs +9 -7
  30. package/dist/cjs/llm/prepareProviderRequest.cjs.map +1 -1
  31. package/dist/cjs/llm/providers.cjs +1 -1
  32. package/dist/cjs/main.cjs +14 -8
  33. package/dist/cjs/messages/alternation.cjs +1 -5
  34. package/dist/cjs/messages/alternation.cjs.map +1 -1
  35. package/dist/cjs/messages/budget.cjs +206 -7
  36. package/dist/cjs/messages/budget.cjs.map +1 -1
  37. package/dist/cjs/messages/cache.cjs +3 -9
  38. package/dist/cjs/messages/cache.cjs.map +1 -1
  39. package/dist/cjs/messages/core.cjs +26 -38
  40. package/dist/cjs/messages/core.cjs.map +1 -1
  41. package/dist/cjs/messages/format.cjs +132 -43
  42. package/dist/cjs/messages/format.cjs.map +1 -1
  43. package/dist/cjs/messages/index.cjs +2 -0
  44. package/dist/cjs/messages/prune.cjs +14 -9
  45. package/dist/cjs/messages/prune.cjs.map +1 -1
  46. package/dist/cjs/messages/reasoningTypes.cjs +21 -0
  47. package/dist/cjs/messages/reasoningTypes.cjs.map +1 -0
  48. package/dist/cjs/messages/recency.cjs +15 -94
  49. package/dist/cjs/messages/recency.cjs.map +1 -1
  50. package/dist/cjs/messages/toolHistoryProjection.cjs +243 -0
  51. package/dist/cjs/messages/toolHistoryProjection.cjs.map +1 -0
  52. package/dist/cjs/messages/toolResultTypes.cjs +145 -4
  53. package/dist/cjs/messages/toolResultTypes.cjs.map +1 -1
  54. package/dist/cjs/run.cjs +22 -10
  55. package/dist/cjs/run.cjs.map +1 -1
  56. package/dist/cjs/session/AgentSession.cjs +7 -5
  57. package/dist/cjs/session/AgentSession.cjs.map +1 -1
  58. package/dist/cjs/session/JsonlSessionStore.cjs +3 -0
  59. package/dist/cjs/session/JsonlSessionStore.cjs.map +1 -1
  60. package/dist/cjs/session/index.cjs +1 -1
  61. package/dist/cjs/session/sessionProjection.cjs +75 -0
  62. package/dist/cjs/session/sessionProjection.cjs.map +1 -0
  63. package/dist/cjs/stream.cjs +20 -19
  64. package/dist/cjs/stream.cjs.map +1 -1
  65. package/dist/cjs/summarization/index.cjs.map +1 -1
  66. package/dist/cjs/summarization/node.cjs +19 -9
  67. package/dist/cjs/summarization/node.cjs.map +1 -1
  68. package/dist/cjs/summarization/shared.cjs +9 -0
  69. package/dist/cjs/summarization/shared.cjs.map +1 -1
  70. package/dist/cjs/tools/ToolNode.cjs +256 -84
  71. package/dist/cjs/tools/ToolNode.cjs.map +1 -1
  72. package/dist/cjs/tools/local/CompileCheckTool.cjs +1 -1
  73. package/dist/cjs/tools/local/LocalCodingTools.cjs +1 -1
  74. package/dist/cjs/tools/subagent/SubagentExecutor.cjs +24 -9
  75. package/dist/cjs/tools/subagent/SubagentExecutor.cjs.map +1 -1
  76. package/dist/cjs/tools/subagent/SubagentReplay.cjs +2 -0
  77. package/dist/cjs/tools/subagent/SubagentReplay.cjs.map +1 -1
  78. package/dist/cjs/tools/toolBatchReplay.cjs +187 -0
  79. package/dist/cjs/tools/toolBatchReplay.cjs.map +1 -0
  80. package/dist/cjs/tools/toolOutputReferences.cjs +12 -0
  81. package/dist/cjs/tools/toolOutputReferences.cjs.map +1 -1
  82. package/dist/cjs/types/hitl.cjs +4 -0
  83. package/dist/cjs/types/hitl.cjs.map +1 -1
  84. package/dist/cjs/utils/events.cjs +13 -0
  85. package/dist/cjs/utils/events.cjs.map +1 -1
  86. package/dist/cjs/utils/tokens.cjs +105 -0
  87. package/dist/cjs/utils/tokens.cjs.map +1 -1
  88. package/dist/esm/agents/AgentContext.mjs +13 -3
  89. package/dist/esm/agents/AgentContext.mjs.map +1 -1
  90. package/dist/esm/graphs/Graph.mjs +178 -57
  91. package/dist/esm/graphs/Graph.mjs.map +1 -1
  92. package/dist/esm/graphs/MultiAgentGraph.mjs +16 -6
  93. package/dist/esm/graphs/MultiAgentGraph.mjs.map +1 -1
  94. package/dist/esm/hitl/approvalReview.mjs +83 -0
  95. package/dist/esm/hitl/approvalReview.mjs.map +1 -0
  96. package/dist/esm/langfuse.mjs +43 -2
  97. package/dist/esm/langfuse.mjs.map +1 -1
  98. package/dist/esm/langfuseToolOutputTracing.mjs +41 -1
  99. package/dist/esm/langfuseToolOutputTracing.mjs.map +1 -1
  100. package/dist/esm/langfuseTraceShaping.mjs +7 -1
  101. package/dist/esm/langfuseTraceShaping.mjs.map +1 -1
  102. package/dist/esm/llm/anthropic/utils/message_inputs.mjs +6 -10
  103. package/dist/esm/llm/anthropic/utils/message_inputs.mjs.map +1 -1
  104. package/dist/esm/llm/bedrock/utils/message_inputs.mjs +2 -7
  105. package/dist/esm/llm/bedrock/utils/message_inputs.mjs.map +1 -1
  106. package/dist/esm/llm/contextPressureMeter.mjs +52 -11
  107. package/dist/esm/llm/contextPressureMeter.mjs.map +1 -1
  108. package/dist/esm/llm/init.mjs +1 -1
  109. package/dist/esm/llm/invoke.mjs +2 -2
  110. package/dist/esm/llm/openai/index.mjs +2 -1
  111. package/dist/esm/llm/openai/index.mjs.map +1 -1
  112. package/dist/esm/llm/openai/utils/index.mjs +5 -4
  113. package/dist/esm/llm/openai/utils/index.mjs.map +1 -1
  114. package/dist/esm/llm/preempt.mjs +2 -1
  115. package/dist/esm/llm/preempt.mjs.map +1 -1
  116. package/dist/esm/llm/prepareProviderRequest.mjs +9 -7
  117. package/dist/esm/llm/prepareProviderRequest.mjs.map +1 -1
  118. package/dist/esm/llm/providers.mjs +1 -1
  119. package/dist/esm/main.mjs +13 -10
  120. package/dist/esm/messages/alternation.mjs +2 -6
  121. package/dist/esm/messages/alternation.mjs.map +1 -1
  122. package/dist/esm/messages/budget.mjs +206 -8
  123. package/dist/esm/messages/budget.mjs.map +1 -1
  124. package/dist/esm/messages/cache.mjs +3 -9
  125. package/dist/esm/messages/cache.mjs.map +1 -1
  126. package/dist/esm/messages/core.mjs +24 -34
  127. package/dist/esm/messages/core.mjs.map +1 -1
  128. package/dist/esm/messages/format.mjs +133 -44
  129. package/dist/esm/messages/format.mjs.map +1 -1
  130. package/dist/esm/messages/index.mjs +2 -0
  131. package/dist/esm/messages/prune.mjs +14 -9
  132. package/dist/esm/messages/prune.mjs.map +1 -1
  133. package/dist/esm/messages/reasoningTypes.mjs +20 -0
  134. package/dist/esm/messages/reasoningTypes.mjs.map +1 -0
  135. package/dist/esm/messages/recency.mjs +16 -95
  136. package/dist/esm/messages/recency.mjs.map +1 -1
  137. package/dist/esm/messages/toolHistoryProjection.mjs +237 -0
  138. package/dist/esm/messages/toolHistoryProjection.mjs.map +1 -0
  139. package/dist/esm/messages/toolResultTypes.mjs +141 -4
  140. package/dist/esm/messages/toolResultTypes.mjs.map +1 -1
  141. package/dist/esm/run.mjs +23 -11
  142. package/dist/esm/run.mjs.map +1 -1
  143. package/dist/esm/session/AgentSession.mjs +7 -5
  144. package/dist/esm/session/AgentSession.mjs.map +1 -1
  145. package/dist/esm/session/JsonlSessionStore.mjs +3 -0
  146. package/dist/esm/session/JsonlSessionStore.mjs.map +1 -1
  147. package/dist/esm/session/index.mjs +1 -1
  148. package/dist/esm/session/sessionProjection.mjs +72 -0
  149. package/dist/esm/session/sessionProjection.mjs.map +1 -0
  150. package/dist/esm/stream.mjs +17 -16
  151. package/dist/esm/stream.mjs.map +1 -1
  152. package/dist/esm/summarization/index.mjs.map +1 -1
  153. package/dist/esm/summarization/node.mjs +20 -10
  154. package/dist/esm/summarization/node.mjs.map +1 -1
  155. package/dist/esm/summarization/shared.mjs +9 -1
  156. package/dist/esm/summarization/shared.mjs.map +1 -1
  157. package/dist/esm/tools/ToolNode.mjs +256 -84
  158. package/dist/esm/tools/ToolNode.mjs.map +1 -1
  159. package/dist/esm/tools/local/CompileCheckTool.mjs +1 -1
  160. package/dist/esm/tools/local/LocalCodingTools.mjs +1 -1
  161. package/dist/esm/tools/subagent/SubagentExecutor.mjs +24 -9
  162. package/dist/esm/tools/subagent/SubagentExecutor.mjs.map +1 -1
  163. package/dist/esm/tools/subagent/SubagentReplay.mjs +1 -1
  164. package/dist/esm/tools/subagent/SubagentReplay.mjs.map +1 -1
  165. package/dist/esm/tools/toolBatchReplay.mjs +176 -0
  166. package/dist/esm/tools/toolBatchReplay.mjs.map +1 -0
  167. package/dist/esm/tools/toolOutputReferences.mjs +12 -0
  168. package/dist/esm/tools/toolOutputReferences.mjs.map +1 -1
  169. package/dist/esm/types/hitl.mjs +4 -1
  170. package/dist/esm/types/hitl.mjs.map +1 -1
  171. package/dist/esm/utils/events.mjs +13 -0
  172. package/dist/esm/utils/events.mjs.map +1 -1
  173. package/dist/esm/utils/tokens.mjs +105 -1
  174. package/dist/esm/utils/tokens.mjs.map +1 -1
  175. package/dist/types/agents/AgentContext.d.ts +13 -1
  176. package/dist/types/graphs/Graph.d.ts +27 -0
  177. package/dist/types/hitl/approvalReview.d.ts +32 -0
  178. package/dist/types/hooks/types.d.ts +4 -2
  179. package/dist/types/index.d.ts +1 -0
  180. package/dist/types/langfuse.d.ts +5 -0
  181. package/dist/types/langfuseToolOutputTracing.d.ts +2 -0
  182. package/dist/types/llm/contextPressureMeter.d.ts +2 -1
  183. package/dist/types/llm/prepareProviderRequest.d.ts +7 -1
  184. package/dist/types/messages/budget.d.ts +15 -10
  185. package/dist/types/messages/core.d.ts +8 -17
  186. package/dist/types/messages/format.d.ts +3 -2
  187. package/dist/types/messages/index.d.ts +1 -1
  188. package/dist/types/messages/prune.d.ts +1 -1
  189. package/dist/types/messages/reasoningTypes.d.ts +7 -0
  190. package/dist/types/messages/recency.d.ts +3 -0
  191. package/dist/types/messages/toolHistoryProjection.d.ts +65 -0
  192. package/dist/types/messages/toolResultTypes.d.ts +31 -1
  193. package/dist/types/session/sessionProjection.d.ts +10 -0
  194. package/dist/types/summarization/index.d.ts +2 -1
  195. package/dist/types/summarization/node.d.ts +1 -0
  196. package/dist/types/summarization/shared.d.ts +12 -0
  197. package/dist/types/tools/ToolNode.d.ts +12 -6
  198. package/dist/types/tools/subagent/SubagentReplay.d.ts +2 -0
  199. package/dist/types/tools/toolBatchReplay.d.ts +54 -0
  200. package/dist/types/tools/toolOutputReferences.d.ts +2 -0
  201. package/dist/types/types/graph.d.ts +20 -0
  202. package/dist/types/types/run.d.ts +4 -0
  203. package/dist/types/types/summarize.d.ts +4 -1
  204. package/dist/types/utils/tokens.d.ts +2 -1
  205. package/package.json +1 -1
  206. package/src/agents/AgentContext.ts +34 -3
  207. package/src/graphs/Graph.ts +465 -147
  208. package/src/graphs/MultiAgentGraph.ts +28 -6
  209. package/src/hitl/approvalReview.ts +209 -0
  210. package/src/hooks/types.ts +4 -1
  211. package/src/index.ts +1 -0
  212. package/src/langfuse.ts +67 -9
  213. package/src/langfuseToolOutputTracing.ts +91 -0
  214. package/src/langfuseTraceShaping.ts +17 -4
  215. package/src/llm/anthropic/utils/message_inputs.ts +5 -10
  216. package/src/llm/bedrock/utils/message_inputs.ts +2 -8
  217. package/src/llm/contextPressureMeter.ts +80 -12
  218. package/src/llm/openai/utils/index.ts +5 -4
  219. package/src/llm/prepareProviderRequest.ts +21 -16
  220. package/src/messages/alternation.ts +2 -15
  221. package/src/messages/budget.ts +439 -23
  222. package/src/messages/cache.ts +8 -9
  223. package/src/messages/core.ts +65 -95
  224. package/src/messages/format.ts +262 -78
  225. package/src/messages/index.ts +1 -1
  226. package/src/messages/prune.ts +31 -12
  227. package/src/messages/reasoningTypes.ts +23 -0
  228. package/src/messages/recency.ts +35 -179
  229. package/src/messages/toolHistoryProjection.ts +460 -0
  230. package/src/messages/toolResultTypes.ts +331 -100
  231. package/src/run.ts +44 -22
  232. package/src/session/AgentSession.ts +33 -16
  233. package/src/session/JsonlSessionStore.ts +6 -0
  234. package/src/session/sessionProjection.ts +126 -0
  235. package/src/stream.ts +14 -19
  236. package/src/summarization/index.ts +2 -0
  237. package/src/summarization/node.ts +62 -13
  238. package/src/summarization/shared.ts +23 -0
  239. package/src/tools/ToolNode.ts +518 -184
  240. package/src/tools/subagent/SubagentExecutor.ts +35 -6
  241. package/src/tools/subagent/SubagentReplay.ts +2 -2
  242. package/src/tools/toolBatchReplay.ts +395 -0
  243. package/src/tools/toolOutputReferences.ts +16 -0
  244. package/src/types/graph.ts +20 -0
  245. package/src/types/run.ts +4 -0
  246. package/src/types/summarize.ts +4 -1
  247. package/src/utils/events.ts +19 -0
  248. package/src/utils/tokens.ts +183 -0
@@ -146,7 +146,10 @@ import {
146
146
  findCallback,
147
147
  type CallbackEntry,
148
148
  } from '@/utils/callbacks';
149
- import { PreparedSubagents, PreparedSubagentError } from '@/tools/preparedSubagents';
149
+ import {
150
+ PreparedSubagents,
151
+ PreparedSubagentError,
152
+ } from '@/tools/preparedSubagents';
150
153
  import { ToolNode as CustomToolNode, toolsCondition } from '@/tools/ToolNode';
151
154
  import { shouldTraceToolNodeForLangfuse } from '@/langfuseToolOutputTracing';
152
155
  import { createLocalCodingToolBundle } from '@/tools/local/LocalCodingTools';
@@ -154,20 +157,30 @@ import { SUBAGENT_REPLAY_CONTROLLER } from '@/tools/subagent/SubagentReplay';
154
157
  import { applyGraphRuntimeConfig } from '@/graphs/applyGraphRuntimeConfig';
155
158
  import { isFadingTier, isInformativeFadingTier } from '@/messages/fading';
156
159
  import { createContextPressureMeter } from '@/llm/contextPressureMeter';
160
+ import { createToolHistoryPreparation } from '@/messages/toolHistoryProjection';
157
161
  import { safeDispatchCustomEvent, emitAgentLog } from '@/utils/events';
158
- import { prepareProviderRequest } from '@/llm/prepareProviderRequest';
162
+ import {
163
+ prepareProviderRequest,
164
+ usesNativeOpenAIResponses,
165
+ } from '@/llm/prepareProviderRequest';
159
166
  import { createCloudflareCodingToolBundle } from '@/tools/cloudflare';
160
167
  import { calculateMaxToolCallInputChars } from '@/utils/truncation';
161
168
  import { prepareToolsForPromptCache } from '@/llm/promptCacheTools';
162
169
  import { providerRequiresStrictAlternation } from '@/llm/providers';
163
170
  import { buildSubagentToolParams } from '@/tools/SubagentTool';
164
171
  import { initializeLangfuseTracing } from '@/instrumentation';
165
- import { shouldTriggerSummarization } from '@/summarization';
172
+ import {
173
+ ManualSummarizationSkippedError,
174
+ shouldTriggerSummarization,
175
+ } from '@/summarization';
166
176
  import { isRunStepResumeState } from '@/tools/runStepResume';
167
177
  import { resolveLocalToolsForBinding } from '@/tools/local';
168
178
  import { createSummarizeNode } from '@/summarization/node';
179
+ import {
180
+ createRemoveAllMessage,
181
+ messagesStateReducer,
182
+ } from '@/messages/reducer';
169
183
  import { getTruncationStopReason } from '@/llm/truncation';
170
- import { messagesStateReducer } from '@/messages/reducer';
171
184
  import { createSchemaOnlyTools } from '@/tools/schema';
172
185
  import { AgentContext } from '@/agents/AgentContext';
173
186
  import { createFakeStreamingLLM } from '@/llm/fake';
@@ -1242,7 +1255,10 @@ export abstract class Graph<
1242
1255
  }> = new Set();
1243
1256
  private _subagentExecutors = new Set<SubagentExecutor>();
1244
1257
  readonly preparedSubagents = new PreparedSubagents();
1245
- protected readonly subagentToolNodes = new Map<string, CustomToolNode<t.BaseGraphState>>();
1258
+ protected readonly subagentToolNodes = new Map<
1259
+ string,
1260
+ CustomToolNode<t.BaseGraphState>
1261
+ >();
1246
1262
  public getOrCreateFileCheckpointer(): t.LocalFileCheckpointer | undefined {
1247
1263
  // Return the cached instance unconditionally if one exists. The
1248
1264
  // toolExecution check below decides whether to *create* a new
@@ -1404,13 +1420,15 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1404
1420
  ): StreamLimitExceededError | PreparedSubagentError | undefined {
1405
1421
  if (
1406
1422
  breakerSignal.aborted &&
1407
- (breakerSignal.reason instanceof StreamLimitExceededError || breakerSignal.reason instanceof PreparedSubagentError)
1423
+ (breakerSignal.reason instanceof StreamLimitExceededError ||
1424
+ breakerSignal.reason instanceof PreparedSubagentError)
1408
1425
  ) {
1409
1426
  return breakerSignal.reason;
1410
1427
  }
1411
1428
  if (
1412
1429
  this.signal?.aborted === true &&
1413
- (this.signal.reason instanceof StreamLimitExceededError || this.signal.reason instanceof PreparedSubagentError)
1430
+ (this.signal.reason instanceof StreamLimitExceededError ||
1431
+ this.signal.reason instanceof PreparedSubagentError)
1414
1432
  ) {
1415
1433
  return this.signal.reason;
1416
1434
  }
@@ -1463,6 +1481,12 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1463
1481
  * cannot suppress an unrelated agent's turn in a multi-agent graph.
1464
1482
  */
1465
1483
  outputTruncatedIncomplete = false;
1484
+ /**
1485
+ * The agent a summarize-only run summarizes with. Set from the first agent
1486
+ * input that opted in; while set, the model step after the summary routes
1487
+ * to END, and a multi-agent workflow compiles to this agent alone.
1488
+ */
1489
+ summarizeOnlyAgentId?: string;
1466
1490
  /**
1467
1491
  * `stopReason` from a `PreemptBoundary` hook that halted the turn.
1468
1492
  *
@@ -1551,6 +1575,9 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1551
1575
  indexTokenCountMap,
1552
1576
  toolExecution
1553
1577
  );
1578
+ if (agentConfig.summarizeOnly === true) {
1579
+ this.summarizeOnlyAgentId ??= agentContext.agentId;
1580
+ }
1554
1581
  if (calibrationRatio != null && calibrationRatio > 0) {
1555
1582
  agentContext.calibrationRatio = calibrationRatio;
1556
1583
  }
@@ -1573,7 +1600,33 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1573
1600
  this.agentContexts.set(agentConfig.agentId, agentContext);
1574
1601
  }
1575
1602
 
1576
- this.defaultAgentId = agents[0].agentId;
1603
+ /** The run's agent: root trace identity and Langfuse routing follow it,
1604
+ * so a summarize-only run is attributed to the agent that summarizes. */
1605
+ this.defaultAgentId = this.summarizeOnlyAgentId ?? agents[0].agentId;
1606
+ }
1607
+
1608
+ /**
1609
+ * A compaction applies its remove-all inside the agent subgraph, so the
1610
+ * subgraph's result carries only the retained tail, and an outer reducer
1611
+ * would merge that tail back into the history it already holds. Re-issuing
1612
+ * the remove-all across the boundary keeps a checkpointed outer state as
1613
+ * compacted as the subgraph's. A run that produced no summary changed
1614
+ * nothing and passes through.
1615
+ */
1616
+ protected propagateManualCompaction<
1617
+ S extends { messages: BaseMessage[]; manualSummary?: string },
1618
+ >(result: S): S {
1619
+ if (
1620
+ this.summarizeOnlyAgentId == null ||
1621
+ result.manualSummary == null ||
1622
+ result.manualSummary.length === 0
1623
+ ) {
1624
+ return result;
1625
+ }
1626
+ return {
1627
+ ...result,
1628
+ messages: [createRemoveAllMessage(), ...result.messages],
1629
+ };
1577
1630
  }
1578
1631
 
1579
1632
  /** Rotates the Langfuse identities that must never cross fresh executions. */
@@ -1957,8 +2010,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
1957
2010
  this.nextContentIndex = state.nextIndex;
1958
2011
  this.runStepStateRevision = state.revision;
1959
2012
  this.stopContinuationCount = state.stopContinuationCount ?? 0;
1960
- this.stopContinuationExecutionId =
1961
- state.stopContinuationExecutionId ?? '';
2013
+ this.stopContinuationExecutionId = state.stopContinuationExecutionId ?? '';
1962
2014
  this.streamSegment = state.streamSegment ?? 0;
1963
2015
  for (const { toolCallId, stepId } of state.toolCallSteps) {
1964
2016
  this.toolCallStepIds.set(toolCallId, stepId);
@@ -3086,6 +3138,79 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3086
3138
  });
3087
3139
  }
3088
3140
 
3141
+ /**
3142
+ * The two model steps of a summarize-only run, neither of which calls the
3143
+ * model. The first requests the summary outright — no trigger consulted,
3144
+ * because the user asked for it — and the second, entered after the
3145
+ * summarize node, dispatches the post-summary usage snapshot (the host's
3146
+ * gauge and summary baseline read it) and ends the run. Returns `null`
3147
+ * for every other run so the ordinary step proceeds.
3148
+ */
3149
+ private async summarizeOnlyStep({
3150
+ agentContext,
3151
+ agentId,
3152
+ messageCount,
3153
+ contextUsage,
3154
+ config,
3155
+ }: {
3156
+ agentContext: AgentContext;
3157
+ agentId: string;
3158
+ messageCount: number;
3159
+ contextUsage: t.ContextUsageEvent | null;
3160
+ config: RunnableConfig;
3161
+ }): Promise<Partial<t.AgentSubgraphState> | null> {
3162
+ if (agentContext.summarizeOnly !== true) {
3163
+ return null;
3164
+ }
3165
+ const meta = { runId: this.runId, agentId };
3166
+ if (agentContext.claimManualSummarization()) {
3167
+ if (agentContext.summarizationEnabled !== true) {
3168
+ throw new ManualSummarizationSkippedError(
3169
+ 'disabled',
3170
+ 'Compaction skipped: summarization is not enabled for this agent'
3171
+ );
3172
+ }
3173
+ emitAgentLog(
3174
+ config,
3175
+ 'info',
3176
+ 'graph',
3177
+ 'Manual summarization requested',
3178
+ {
3179
+ totalMessages: messageCount,
3180
+ summaryVersion: agentContext.summaryVersion + 1,
3181
+ },
3182
+ meta
3183
+ );
3184
+ agentContext.markSummarizationTriggered(messageCount);
3185
+ return {
3186
+ summarizationRequest: {
3187
+ remainingContextTokens: contextUsage?.remainingContextTokens ?? 0,
3188
+ agentId: agentId || agentContext.agentId,
3189
+ reason: 'manual',
3190
+ },
3191
+ /** A checkpointed thread may still hold the previous compaction's
3192
+ * summary; only the one this run produces may be reported. */
3193
+ manualSummary: '',
3194
+ };
3195
+ }
3196
+ if (contextUsage != null) {
3197
+ await safeDispatchCustomEvent(
3198
+ GraphEvents.ON_CONTEXT_USAGE,
3199
+ contextUsage,
3200
+ config
3201
+ );
3202
+ }
3203
+ emitAgentLog(
3204
+ config,
3205
+ 'debug',
3206
+ 'graph',
3207
+ 'Summarize-only run complete — ending without a model call',
3208
+ { summaryVersion: agentContext.summaryVersion },
3209
+ meta
3210
+ );
3211
+ return { messages: [] };
3212
+ }
3213
+
3089
3214
  createCallModel(agentId = 'default') {
3090
3215
  return async (
3091
3216
  state: t.AgentSubgraphState,
@@ -3143,35 +3268,6 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3143
3268
  agentContext.markToolsAsDiscovered(discoveredNames);
3144
3269
  }
3145
3270
 
3146
- /**
3147
- * Anthropic prompt-cache breakpoint on the tool definitions.
3148
- *
3149
- * Without this, the (often static) tool inventory shows up as
3150
- * fresh input on every turn — measured at ~28k tokens/turn for
3151
- * the local engine's coding-tool bundle, dominating per-turn
3152
- * cost even when message-level caching is on.
3153
- *
3154
- * Strategy: partition tools into [static, deferred] and stamp
3155
- * `cache_control: ephemeral` on the last static tool.
3156
- * Discovered deferred tools that arrive across turns sit *after*
3157
- * the breakpoint and don't invalidate the prefix.
3158
- */
3159
- const toolsForBinding = this.getPreparedToolsForBinding(agentContext);
3160
-
3161
- let model =
3162
- this.overrideModel ??
3163
- initializeModel({
3164
- tools: toolsForBinding,
3165
- provider: agentContext.provider,
3166
- clientOptions: agentContext.clientOptions,
3167
- });
3168
-
3169
- if (agentContext.systemRunnable) {
3170
- model = agentContext.systemRunnable
3171
- .pipe(model as Runnable)
3172
- .withConfig({ runName: AGENT_MODEL_CALL_RUN_NAME });
3173
- }
3174
-
3175
3271
  if (agentContext.tokenCalculationPromise) {
3176
3272
  await agentContext.tokenCalculationPromise;
3177
3273
  }
@@ -3309,6 +3405,17 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3309
3405
  };
3310
3406
  syncBudgetDerivedFields(contextUsage);
3311
3407
 
3408
+ const summarizeOnlyStep = await this.summarizeOnlyStep({
3409
+ agentContext,
3410
+ agentId,
3411
+ messageCount: messages.length,
3412
+ contextUsage,
3413
+ config,
3414
+ });
3415
+ if (summarizeOnlyStep != null) {
3416
+ return summarizeOnlyStep;
3417
+ }
3418
+
3312
3419
  const hasPrunedMessages =
3313
3420
  agentContext.summarizationEnabled === true &&
3314
3421
  Array.isArray(messagesToRefine) &&
@@ -3381,9 +3488,57 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3381
3488
  );
3382
3489
  }
3383
3490
  }
3491
+ } else {
3492
+ /** No pruner (no token counter or budget): the summarize-only run
3493
+ * still has to request its summary and stop, just without usage. */
3494
+ const summarizeOnlyStep = await this.summarizeOnlyStep({
3495
+ agentContext,
3496
+ agentId,
3497
+ messageCount: messages.length,
3498
+ contextUsage: null,
3499
+ config,
3500
+ });
3501
+ if (summarizeOnlyStep != null) {
3502
+ return summarizeOnlyStep;
3503
+ }
3504
+ }
3505
+
3506
+ /**
3507
+ * Anthropic prompt-cache breakpoint on the tool definitions.
3508
+ *
3509
+ * Without this, the (often static) tool inventory shows up as
3510
+ * fresh input on every turn — measured at ~28k tokens/turn for
3511
+ * the local engine's coding-tool bundle, dominating per-turn
3512
+ * cost even when message-level caching is on.
3513
+ *
3514
+ * Strategy: partition tools into [static, deferred] and stamp
3515
+ * `cache_control: ephemeral` on the last static tool.
3516
+ * Discovered deferred tools that arrive across turns sit *after*
3517
+ * the breakpoint and don't invalidate the prefix.
3518
+ *
3519
+ * Built only once a model call is certain: a summarize-only step
3520
+ * returns above without one, and must not fail on a primary model
3521
+ * it never invokes.
3522
+ */
3523
+ const toolsForBinding = this.getPreparedToolsForBinding(agentContext);
3524
+
3525
+ let model =
3526
+ this.overrideModel ??
3527
+ initializeModel({
3528
+ tools: toolsForBinding,
3529
+ provider: agentContext.provider,
3530
+ clientOptions: agentContext.clientOptions,
3531
+ });
3532
+
3533
+ if (agentContext.systemRunnable) {
3534
+ model = agentContext.systemRunnable
3535
+ .pipe(model as Runnable)
3536
+ .withConfig({ runName: AGENT_MODEL_CALL_RUN_NAME });
3384
3537
  }
3385
3538
 
3386
3539
  let finalMessages = messagesToUse;
3540
+ /** Primary wire shaping is lossy; each fallback starts from pruned source history. */
3541
+ const fallbackBaseMessages = messagesToUse;
3387
3542
  /**
3388
3543
  * Keep the pruner's provider-grounded aggregate as the authoritative
3389
3544
  * baseline, then attribute it across retained messages. Provider
@@ -3392,6 +3547,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3392
3547
  * the expansion. Exact counts are memoized across repeated projections.
3393
3548
  */
3394
3549
  const contextPressure = createContextPressureMeter({
3550
+ provider: agentContext.provider,
3395
3551
  tokenCounter: agentContext.tokenCounter,
3396
3552
  tokenCountCache: agentContext.contextPressureTokenCounts,
3397
3553
  sourceMessages: messages,
@@ -3402,6 +3558,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3402
3558
  calibrationRatio: agentContext.calibrationRatio,
3403
3559
  });
3404
3560
  const trackProviderMessageOrigins = contextPressure.trackProjection;
3561
+ const toolHistory = createToolHistoryPreparation();
3405
3562
 
3406
3563
  if (agentContext.useLegacyContent) {
3407
3564
  const before = finalMessages;
@@ -3419,7 +3576,8 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3419
3576
  beforeToolInputProjection,
3420
3577
  projectToolMessagesForProvider(
3421
3578
  beforeToolInputProjection,
3422
- calculateMaxToolCallInputChars(agentContext.maxContextTokens)
3579
+ calculateMaxToolCallInputChars(agentContext.maxContextTokens),
3580
+ agentContext.provider
3423
3581
  )
3424
3582
  );
3425
3583
 
@@ -3496,12 +3654,24 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3496
3654
  };
3497
3655
 
3498
3656
  const applyProviderMessageTransforms = (
3499
- candidate: BaseMessage[]
3657
+ candidate: BaseMessage[],
3658
+ serving?: {
3659
+ provider: t.ProviderName;
3660
+ clientOptions?: t.ClientOptions;
3661
+ model: t.ChatModel;
3662
+ config?: RunnableConfig;
3663
+ history: ReturnType<typeof createToolHistoryPreparation>;
3664
+ }
3500
3665
  ): BaseMessage[] => {
3666
+ const provider = serving?.provider ?? agentContext.provider;
3667
+ const clientOptions =
3668
+ serving == null ? agentContext.clientOptions : serving.clientOptions;
3669
+ const callConfig = serving == null ? config : serving.config;
3670
+ const history = serving?.history ?? toolHistory;
3671
+ const servingModel =
3672
+ serving?.model ?? ((this.overrideModel ?? model) as t.ChatModel);
3501
3673
  let transformed = candidate;
3502
- if (
3503
- isThinkingEnabled(agentContext.provider, agentContext.clientOptions)
3504
- ) {
3674
+ if (isThinkingEnabled(provider, clientOptions)) {
3505
3675
  /**
3506
3676
  * Current-run AI messages may validly omit a thinking block. The
3507
3677
  * boundary prevents them from being mistaken for foreign history.
@@ -3511,9 +3681,10 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3511
3681
  before,
3512
3682
  ensureThinkingBlockInMessages(
3513
3683
  before,
3514
- agentContext.provider,
3515
- config,
3516
- this.startIndex
3684
+ provider,
3685
+ callConfig,
3686
+ this.startIndex,
3687
+ history
3517
3688
  )
3518
3689
  );
3519
3690
  }
@@ -3526,7 +3697,12 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3526
3697
  const before = transformed;
3527
3698
  transformed = trackProviderMessageOrigins(
3528
3699
  before,
3529
- foldToolBlocksForToollessAgent(before, config)
3700
+ foldToolBlocksForToollessAgent(
3701
+ before,
3702
+ callConfig,
3703
+ usesNativeOpenAIResponses(servingModel, provider, callConfig),
3704
+ history
3705
+ )
3530
3706
  );
3531
3707
  if (agentContext.useLegacyContent) {
3532
3708
  const beforeLegacyFormat = transformed;
@@ -3545,12 +3721,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3545
3721
  * tolerant fallback and adds it for a Claude fallback behind a
3546
3722
  * tolerant primary.
3547
3723
  */
3548
- if (
3549
- isAnthropicLike(
3550
- agentContext.provider,
3551
- agentContext.clientOptions as { model?: string }
3552
- )
3553
- ) {
3724
+ if (isAnthropicLike(provider, clientOptions as { model?: string })) {
3554
3725
  const before = transformed;
3555
3726
  transformed = trackProviderMessageOrigins(
3556
3727
  before,
@@ -3573,7 +3744,8 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3573
3744
  );
3574
3745
 
3575
3746
  const compactSyntheticProviderContext = (
3576
- candidate: BaseMessage[]
3747
+ candidate: BaseMessage[],
3748
+ measure = measureProviderPayload
3577
3749
  ): BaseMessage[] => {
3578
3750
  const synthetic: Array<{
3579
3751
  index: number;
@@ -3611,7 +3783,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3611
3783
  };
3612
3784
 
3613
3785
  let best = buildCandidate(0);
3614
- if (!measureProviderPayload(best).fits) {
3786
+ if (!measure(best).fits) {
3615
3787
  return candidate;
3616
3788
  }
3617
3789
  let low = 0;
@@ -3619,7 +3791,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3619
3791
  for (let i = 0; i < 12; i++) {
3620
3792
  const scale = (low + high) / 2;
3621
3793
  const attempt = buildCandidate(scale);
3622
- if (measureProviderPayload(attempt).fits) {
3794
+ if (measure(attempt).fits) {
3623
3795
  best = attempt;
3624
3796
  low = scale;
3625
3797
  } else {
@@ -3629,27 +3801,83 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3629
3801
  return best;
3630
3802
  };
3631
3803
 
3632
- let artifactBaseMessages: BaseMessage[] | undefined;
3633
- if (lastMessageY instanceof ToolMessage) {
3634
- let artifactCandidate = finalMessages;
3635
- if (anthropicLike) {
3636
- artifactCandidate = trackProviderMessageOrigins(
3637
- finalMessages,
3638
- projectAnthropicArtifactContent(
3639
- finalMessages,
3640
- maxProviderToolResultChars
3641
- )
3804
+ const projectServingArtifacts = (
3805
+ candidate: BaseMessage[],
3806
+ provider: t.ProviderName,
3807
+ clientOptions: t.ClientOptions | undefined,
3808
+ maxChars: number
3809
+ ): BaseMessage[] => {
3810
+ if (!(candidate[candidate.length - 1] instanceof ToolMessage)) {
3811
+ return candidate;
3812
+ }
3813
+ if (isAnthropicLike(provider, clientOptions as { model?: string })) {
3814
+ return trackProviderMessageOrigins(
3815
+ candidate,
3816
+ projectAnthropicArtifactContent(candidate, maxChars)
3642
3817
  );
3643
- } else if (
3644
- (isOpenAILike(agentContext.provider) &&
3645
- agentContext.provider !== Providers.DEEPSEEK) ||
3646
- isGoogleLike(agentContext.provider)
3818
+ }
3819
+ if (
3820
+ (isOpenAILike(provider) && provider !== Providers.DEEPSEEK) ||
3821
+ isGoogleLike(provider)
3647
3822
  ) {
3648
- artifactCandidate = trackProviderMessageOrigins(
3649
- finalMessages,
3650
- projectArtifactPayload(finalMessages, maxProviderToolResultChars)
3823
+ return trackProviderMessageOrigins(
3824
+ candidate,
3825
+ projectArtifactPayload(candidate, maxChars)
3651
3826
  );
3652
3827
  }
3828
+ return candidate;
3829
+ };
3830
+
3831
+ const applyServingTailCache = (
3832
+ candidate: BaseMessage[],
3833
+ provider: t.ProviderName,
3834
+ clientOptions: t.ClientOptions | undefined,
3835
+ systemOwnsTail = false
3836
+ ): BaseMessage[] => {
3837
+ if (provider === Providers.BEDROCK) {
3838
+ const options = clientOptions as
3839
+ | t.BedrockAnthropicClientOptions
3840
+ | undefined;
3841
+ return options?.promptCache === true
3842
+ ? trackProviderMessageOrigins(
3843
+ candidate,
3844
+ addBedrockTailCacheControl(
3845
+ candidate,
3846
+ resolveBedrockPromptCacheTtl(
3847
+ options.promptCacheTtl,
3848
+ options.model
3849
+ )
3850
+ )
3851
+ )
3852
+ : candidate;
3853
+ }
3854
+ if (
3855
+ systemOwnsTail ||
3856
+ (provider !== Providers.ANTHROPIC &&
3857
+ provider !== Providers.OPENROUTER)
3858
+ ) {
3859
+ return candidate;
3860
+ }
3861
+ const options = clientOptions as t.AnthropicClientOptions | undefined;
3862
+ return options?.promptCache === true
3863
+ ? trackProviderMessageOrigins(
3864
+ candidate,
3865
+ addTailCacheControl(
3866
+ candidate,
3867
+ resolvePromptCacheTtl(options.promptCacheTtl)
3868
+ )
3869
+ )
3870
+ : candidate;
3871
+ };
3872
+
3873
+ let artifactBaseMessages: BaseMessage[] | undefined;
3874
+ if (lastMessageY instanceof ToolMessage) {
3875
+ const artifactCandidate = projectServingArtifacts(
3876
+ finalMessages,
3877
+ agentContext.provider,
3878
+ agentContext.clientOptions,
3879
+ maxProviderToolResultChars
3880
+ );
3653
3881
 
3654
3882
  if (artifactCandidate !== finalMessages) {
3655
3883
  const projection = measureProviderPayload(artifactCandidate);
@@ -3843,51 +4071,14 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3843
4071
  // call (zero message caching). Anchoring on the final message list keeps
3844
4072
  // the marker on a block that actually ships. The system-runnable path
3845
4073
  // adds its body marker in AgentContext, so this node skips it there.
3846
- if (
3847
- (anthropicPromptCacheEnabled || openRouterPromptCacheEnabled) &&
3848
- !agentContext.systemRunnable
3849
- ) {
3850
- const beforeCacheControl = finalMessages;
3851
- finalMessages = trackProviderMessageOrigins(
3852
- beforeCacheControl,
3853
- addTailCacheControl<BaseMessage>(
3854
- beforeCacheControl,
3855
- resolvePromptCacheTtl(
3856
- anthropicPromptCacheEnabled
3857
- ? (
3858
- agentContext.clientOptions as
3859
- | t.AnthropicClientOptions
3860
- | undefined
3861
- )?.promptCacheTtl
3862
- : (
3863
- agentContext.clientOptions as
3864
- | t.ProviderOptionsMap[Providers.OPENROUTER]
3865
- | undefined
3866
- )?.promptCacheTtl
3867
- )
3868
- )
3869
- );
3870
- } else if (bedrockPromptCacheEnabled) {
3871
- const bedrockOptions = agentContext.clientOptions as
3872
- | t.BedrockAnthropicClientOptions
3873
- | undefined;
3874
- // Non-Claude models (Nova) reject the extended 1h TTL, so resolve it
3875
- // against the model — message/system caching stays on, clamped to 5m.
3876
- const beforeCacheControl = finalMessages;
3877
- finalMessages = trackProviderMessageOrigins(
3878
- beforeCacheControl,
3879
- addBedrockTailCacheControl<BaseMessage>(
3880
- beforeCacheControl,
3881
- resolveBedrockPromptCacheTtl(
3882
- bedrockOptions?.promptCacheTtl,
3883
- (bedrockOptions as { model?: string } | undefined)?.model
3884
- )
3885
- )
3886
- );
3887
- }
4074
+ finalMessages = applyServingTailCache(
4075
+ finalMessages,
4076
+ agentContext.provider,
4077
+ agentContext.clientOptions,
4078
+ agentContext.systemRunnable != null
4079
+ );
3888
4080
 
3889
- const fallbackBaseMessages = finalMessages;
3890
- const beforeFinalProviderProjection = fallbackBaseMessages;
4081
+ const beforeFinalProviderProjection = finalMessages;
3891
4082
  const preparedRequest = prepareProviderRequest({
3892
4083
  model: (this.overrideModel ?? model) as t.ChatModel,
3893
4084
  messages: beforeFinalProviderProjection,
@@ -3895,6 +4086,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
3895
4086
  context: this,
3896
4087
  config,
3897
4088
  maxToolResultChars: maxProviderToolResultChars,
4089
+ toolHistory,
3898
4090
  measure: (preparedMessages) =>
3899
4091
  measureProviderPayload(
3900
4092
  trackProviderMessageOrigins(
@@ -4006,7 +4198,18 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4006
4198
  finalProjection.projectedMessageTokens
4007
4199
  );
4008
4200
  }
4009
- syncBudgetDerivedFields(contextUsage);
4201
+ contextUsage.breakdown.toolMessageTokens =
4202
+ finalProjection.toolMessageTokens;
4203
+ contextUsage.breakdown.toolMessageTokenCounts =
4204
+ finalProjection.toolMessageTokenCounts;
4205
+ syncBudgetDerivedFields(
4206
+ contextUsage,
4207
+ undefined,
4208
+ undefined,
4209
+ config,
4210
+ agentContext.provider,
4211
+ finalProjection.toolMessageUsageError
4212
+ );
4010
4213
  /** Awaited so async host handlers receive the pre-invoke snapshot
4011
4214
  * before any model deltas are emitted */
4012
4215
  await safeDispatchCustomEvent(
@@ -4188,6 +4391,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4188
4391
  const canSummarizeOverflow =
4189
4392
  agentContext.summarizationEnabled === true &&
4190
4393
  splitAtRecencyBoundary(messages, {
4394
+ provider: agentContext.provider,
4191
4395
  turns:
4192
4396
  agentContext.summarizationConfig?.retainRecent?.turns ??
4193
4397
  DEFAULT_RETAIN_RECENT_TURNS,
@@ -4344,9 +4548,11 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4344
4548
  model: fallbackModel,
4345
4549
  messages: fallbackMessages,
4346
4550
  provider: fallbackProvider,
4551
+ clientOptions: fallbackClientOptions,
4347
4552
  maxContextTokens: fallbackMaxContextTokens,
4348
4553
  config: fallbackConfig,
4349
4554
  }) => {
4555
+ const fallbackToolHistory = createToolHistoryPreparation();
4350
4556
  const fallbackToolResultChars =
4351
4557
  agentContext.maxToolResultChars ??
4352
4558
  calculateMaxToolResultChars(
@@ -4360,31 +4566,112 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4360
4566
  primaryContextBudget ?? fallbackMaxContextTokens,
4361
4567
  fallbackMaxContextTokens
4362
4568
  );
4363
- const preparedFallbackRequest = prepareProviderRequest({
4364
- model: fallbackModel,
4365
- messages: fallbackMessages,
4366
- provider: fallbackProvider,
4367
- context: this,
4368
- config: fallbackConfig,
4369
- maxToolResultChars: fallbackToolResultChars,
4370
- measure: (preparedMessages) =>
4371
- measureProviderPayload(
4372
- trackProviderMessageOrigins(
4373
- fallbackMessages,
4374
- preparedMessages
4375
- ),
4376
- {
4377
- contextBudget: fallbackContextBudget,
4378
- forceRawRecount: true,
4379
- }
4380
- ),
4381
- });
4382
- const projection =
4383
- preparedFallbackRequest.measurement ??
4384
- measureProviderPayload(preparedFallbackRequest.messages, {
4569
+ const measureFallback = (
4570
+ candidate: BaseMessage[]
4571
+ ): ReturnType<typeof measureProviderPayload> =>
4572
+ measureProviderPayload(candidate, {
4385
4573
  contextBudget: fallbackContextBudget,
4386
4574
  forceRawRecount: true,
4387
4575
  });
4576
+ const source = agentContext.useLegacyContent
4577
+ ? trackProviderMessageOrigins(
4578
+ fallbackMessages,
4579
+ formatContentStrings(fallbackMessages)
4580
+ )
4581
+ : fallbackMessages;
4582
+ const boundedSource = trackProviderMessageOrigins(
4583
+ source,
4584
+ projectToolMessagesForProvider(
4585
+ source,
4586
+ calculateMaxToolCallInputChars(fallbackMaxContextTokens),
4587
+ fallbackProvider
4588
+ )
4589
+ );
4590
+ const artifactSource = projectServingArtifacts(
4591
+ boundedSource,
4592
+ fallbackProvider,
4593
+ fallbackClientOptions,
4594
+ fallbackToolResultChars
4595
+ );
4596
+ const transformFallback = (
4597
+ candidate: BaseMessage[]
4598
+ ): BaseMessage[] =>
4599
+ applyProviderMessageTransforms(candidate, {
4600
+ provider: fallbackProvider,
4601
+ clientOptions: fallbackClientOptions,
4602
+ model: fallbackModel,
4603
+ config: fallbackConfig,
4604
+ history: fallbackToolHistory,
4605
+ });
4606
+ const prepareFallback = (
4607
+ candidate: BaseMessage[]
4608
+ ): {
4609
+ request: ReturnType<typeof prepareProviderRequest>;
4610
+ projection: ReturnType<typeof measureProviderPayload>;
4611
+ } => {
4612
+ const sanitized = isAnthropicLike(
4613
+ fallbackProvider,
4614
+ fallbackClientOptions as { model?: string }
4615
+ )
4616
+ ? trackProviderMessageOrigins(
4617
+ candidate,
4618
+ sanitizeOrphanToolBlocks(
4619
+ candidate,
4620
+ (original, clone) =>
4621
+ contextPressure.trackClone(original, clone)
4622
+ )
4623
+ )
4624
+ : candidate;
4625
+ const cached = applyServingTailCache(
4626
+ sanitized,
4627
+ fallbackProvider,
4628
+ fallbackClientOptions
4629
+ );
4630
+ let projection: ReturnType<typeof measureProviderPayload> | undefined;
4631
+ const request = prepareProviderRequest({
4632
+ model: fallbackModel,
4633
+ messages: cached,
4634
+ provider: fallbackProvider,
4635
+ context: this,
4636
+ config: fallbackConfig,
4637
+ maxToolResultChars: fallbackToolResultChars,
4638
+ toolHistory: fallbackToolHistory,
4639
+ measure: (prepared) => {
4640
+ projection = measureFallback(
4641
+ trackProviderMessageOrigins(cached, prepared)
4642
+ );
4643
+ return projection;
4644
+ },
4645
+ });
4646
+ if (projection == null) {
4647
+ throw new Error('Fallback preparation did not measure its payload');
4648
+ }
4649
+ return { request, projection };
4650
+ };
4651
+ let servingFallbackMessages =
4652
+ transformFallback(artifactSource);
4653
+ let preparedFallbackRequest = prepareFallback(
4654
+ servingFallbackMessages
4655
+ );
4656
+ let projection = preparedFallbackRequest.projection;
4657
+ if (!projection.fits && artifactSource !== boundedSource) {
4658
+ servingFallbackMessages = transformFallback(boundedSource);
4659
+ preparedFallbackRequest = prepareFallback(
4660
+ servingFallbackMessages
4661
+ );
4662
+ projection = preparedFallbackRequest.projection;
4663
+ }
4664
+ if (!projection.fits) {
4665
+ const compacted = compactSyntheticProviderContext(
4666
+ servingFallbackMessages,
4667
+ (candidate) =>
4668
+ prepareFallback(candidate).projection
4669
+ );
4670
+ if (compacted !== servingFallbackMessages) {
4671
+ preparedFallbackRequest = prepareFallback(compacted);
4672
+ projection = preparedFallbackRequest.projection;
4673
+ }
4674
+ }
4388
4675
  if (!projection.fits) {
4389
4676
  throw createProviderPayloadOverflowError({
4390
4677
  projection,
@@ -4392,7 +4679,7 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
4392
4679
  info: 'Fallback provider message formatting exceeded the context budget before invocation.',
4393
4680
  });
4394
4681
  }
4395
- return preparedFallbackRequest;
4682
+ return preparedFallbackRequest.request;
4396
4683
  },
4397
4684
  })
4398
4685
  );
@@ -5129,7 +5416,17 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
5129
5416
  this.config = config;
5130
5417
  this.restoreRunStepResumeState(state.runStepState);
5131
5418
  const result = await invoke();
5132
- return { ...result, runStepState: this.createRunStepResumeState() };
5419
+ /** An ordinary run on a checkpointed thread inherits the last
5420
+ * compaction's summary in state; it is not this run's output. */
5421
+ const clearsManualSummary =
5422
+ this.summarizeOnlyAgentId == null &&
5423
+ typeof state.manualSummary === 'string' &&
5424
+ state.manualSummary.length > 0;
5425
+ return {
5426
+ ...result,
5427
+ ...(clearsManualSummary ? { manualSummary: '' } : {}),
5428
+ runStepState: this.createRunStepResumeState(),
5429
+ };
5133
5430
  };
5134
5431
 
5135
5432
  const routeMessage = (
@@ -5148,6 +5445,11 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
5148
5445
  if (state.summarizationRequest != null) {
5149
5446
  return summarizeNode;
5150
5447
  }
5448
+ /** A summarize-only run never calls the model, so the step after the
5449
+ * summary has nothing to route: the summary itself is the result. */
5450
+ if (this.summarizeOnlyAgentId != null) {
5451
+ return END;
5452
+ }
5151
5453
  const decision = toolsCondition(
5152
5454
  state as t.BaseGraphState,
5153
5455
  toolNode,
@@ -5186,6 +5488,10 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
5186
5488
  ) => b,
5187
5489
  default: () => undefined,
5188
5490
  }),
5491
+ manualSummary: Annotation<string | undefined>({
5492
+ reducer: (_: string | undefined, b: string | undefined) => b,
5493
+ default: () => undefined,
5494
+ }),
5189
5495
  runStepState: this.createRunStepStateAnnotation(),
5190
5496
  });
5191
5497
 
@@ -5395,15 +5701,27 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
5395
5701
  },
5396
5702
  default: () => [],
5397
5703
  }),
5704
+ /** Surfaced from the agent subgraph on a summarize-only run. */
5705
+ manualSummary: Annotation<string | undefined>({
5706
+ reducer: (_: string | undefined, b: string | undefined) => b,
5707
+ default: () => undefined,
5708
+ }),
5398
5709
  runStepState: this.createRunStepStateAnnotation(),
5399
5710
  });
5711
+ const compactingAgentNode = async (
5712
+ state: t.AgentSubgraphState,
5713
+ config?: RunnableConfig
5714
+ ): Promise<Partial<t.AgentSubgraphState>> =>
5715
+ this.propagateManualCompaction(await agentNode.invoke(state, config));
5400
5716
  const workflow = new StateGraph(StateAnnotation)
5401
5717
  .addNode(
5402
5718
  this.defaultAgentId,
5403
- agentNode as Runnable<
5404
- t.AgentSubgraphState,
5405
- Partial<t.AgentSubgraphState>
5406
- >,
5719
+ this.summarizeOnlyAgentId != null
5720
+ ? compactingAgentNode
5721
+ : (agentNode as Runnable<
5722
+ t.AgentSubgraphState,
5723
+ Partial<t.AgentSubgraphState>
5724
+ >),
5407
5725
  { ends: [END] }
5408
5726
  )
5409
5727
  .addEdge(START, this.defaultAgentId);