@librechat/agents 3.3.7 → 3.3.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (168) hide show
  1. package/dist/cjs/graphs/Graph.cjs +47 -13
  2. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  3. package/dist/cjs/graphs/MultiAgentGraph.cjs +56 -6
  4. package/dist/cjs/graphs/MultiAgentGraph.cjs.map +1 -1
  5. package/dist/cjs/hitl/askUserQuestion.cjs +3 -2
  6. package/dist/cjs/hitl/askUserQuestion.cjs.map +1 -1
  7. package/dist/cjs/instrumentation.cjs +18 -48
  8. package/dist/cjs/instrumentation.cjs.map +1 -1
  9. package/dist/cjs/langfuse.cjs +174 -29
  10. package/dist/cjs/langfuse.cjs.map +1 -1
  11. package/dist/cjs/langfuseConfig.cjs +12 -0
  12. package/dist/cjs/langfuseConfig.cjs.map +1 -1
  13. package/dist/cjs/langfuseRuntimeContext.cjs +23 -2
  14. package/dist/cjs/langfuseRuntimeContext.cjs.map +1 -1
  15. package/dist/cjs/langfuseRuntimeScope.cjs +39 -8
  16. package/dist/cjs/langfuseRuntimeScope.cjs.map +1 -1
  17. package/dist/cjs/langfuseSpanRegistry.cjs +95 -0
  18. package/dist/cjs/langfuseSpanRegistry.cjs.map +1 -0
  19. package/dist/cjs/langfuseTraceShaping.cjs +121 -4
  20. package/dist/cjs/langfuseTraceShaping.cjs.map +1 -1
  21. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs +39 -15
  22. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs.map +1 -1
  23. package/dist/cjs/llm/bedrock/utils/message_inputs.cjs +25 -5
  24. package/dist/cjs/llm/bedrock/utils/message_inputs.cjs.map +1 -1
  25. package/dist/cjs/llm/init.cjs +3 -3
  26. package/dist/cjs/llm/invoke.cjs +5 -5
  27. package/dist/cjs/llm/openai/index.cjs +1 -1
  28. package/dist/cjs/main.cjs +10 -10
  29. package/dist/cjs/messages/format.cjs +124 -15
  30. package/dist/cjs/messages/format.cjs.map +1 -1
  31. package/dist/cjs/messages/injected.cjs +10 -1
  32. package/dist/cjs/messages/injected.cjs.map +1 -1
  33. package/dist/cjs/messages/prune.cjs +13 -1
  34. package/dist/cjs/messages/prune.cjs.map +1 -1
  35. package/dist/cjs/prompts/activityLabel.cjs +51 -11
  36. package/dist/cjs/prompts/activityLabel.cjs.map +1 -1
  37. package/dist/cjs/run.cjs +54 -24
  38. package/dist/cjs/run.cjs.map +1 -1
  39. package/dist/cjs/session/messageSerialization.cjs +6 -0
  40. package/dist/cjs/session/messageSerialization.cjs.map +1 -1
  41. package/dist/cjs/stream.cjs +21 -10
  42. package/dist/cjs/stream.cjs.map +1 -1
  43. package/dist/cjs/summarization/node.cjs +60 -0
  44. package/dist/cjs/summarization/node.cjs.map +1 -1
  45. package/dist/cjs/tools/ToolNode.cjs +253 -24
  46. package/dist/cjs/tools/ToolNode.cjs.map +1 -1
  47. package/dist/cjs/tools/handlers.cjs +1 -1
  48. package/dist/cjs/tools/search/tool.cjs +1 -1
  49. package/dist/cjs/tools/subagent/SubagentExecutor.cjs +1 -1
  50. package/dist/cjs/utils/index.cjs +2 -2
  51. package/dist/esm/graphs/Graph.mjs +48 -14
  52. package/dist/esm/graphs/Graph.mjs.map +1 -1
  53. package/dist/esm/graphs/MultiAgentGraph.mjs +56 -6
  54. package/dist/esm/graphs/MultiAgentGraph.mjs.map +1 -1
  55. package/dist/esm/hitl/askUserQuestion.mjs +3 -2
  56. package/dist/esm/hitl/askUserQuestion.mjs.map +1 -1
  57. package/dist/esm/instrumentation.mjs +18 -48
  58. package/dist/esm/instrumentation.mjs.map +1 -1
  59. package/dist/esm/langfuse.mjs +176 -28
  60. package/dist/esm/langfuse.mjs.map +1 -1
  61. package/dist/esm/langfuseConfig.mjs +10 -1
  62. package/dist/esm/langfuseConfig.mjs.map +1 -1
  63. package/dist/esm/langfuseRuntimeContext.mjs +21 -3
  64. package/dist/esm/langfuseRuntimeContext.mjs.map +1 -1
  65. package/dist/esm/langfuseRuntimeScope.mjs +39 -10
  66. package/dist/esm/langfuseRuntimeScope.mjs.map +1 -1
  67. package/dist/esm/langfuseSpanRegistry.mjs +91 -0
  68. package/dist/esm/langfuseSpanRegistry.mjs.map +1 -0
  69. package/dist/esm/langfuseTraceShaping.mjs +121 -4
  70. package/dist/esm/langfuseTraceShaping.mjs.map +1 -1
  71. package/dist/esm/llm/anthropic/utils/message_inputs.mjs +39 -15
  72. package/dist/esm/llm/anthropic/utils/message_inputs.mjs.map +1 -1
  73. package/dist/esm/llm/bedrock/utils/message_inputs.mjs +25 -5
  74. package/dist/esm/llm/bedrock/utils/message_inputs.mjs.map +1 -1
  75. package/dist/esm/llm/init.mjs +2 -2
  76. package/dist/esm/llm/invoke.mjs +5 -5
  77. package/dist/esm/llm/openai/index.mjs +1 -1
  78. package/dist/esm/main.mjs +8 -8
  79. package/dist/esm/messages/format.mjs +124 -15
  80. package/dist/esm/messages/format.mjs.map +1 -1
  81. package/dist/esm/messages/injected.mjs +10 -1
  82. package/dist/esm/messages/injected.mjs.map +1 -1
  83. package/dist/esm/messages/prune.mjs +13 -1
  84. package/dist/esm/messages/prune.mjs.map +1 -1
  85. package/dist/esm/prompts/activityLabel.mjs +51 -11
  86. package/dist/esm/prompts/activityLabel.mjs.map +1 -1
  87. package/dist/esm/run.mjs +54 -24
  88. package/dist/esm/run.mjs.map +1 -1
  89. package/dist/esm/session/messageSerialization.mjs +6 -0
  90. package/dist/esm/session/messageSerialization.mjs.map +1 -1
  91. package/dist/esm/stream.mjs +21 -10
  92. package/dist/esm/stream.mjs.map +1 -1
  93. package/dist/esm/summarization/node.mjs +60 -0
  94. package/dist/esm/summarization/node.mjs.map +1 -1
  95. package/dist/esm/tools/ToolNode.mjs +254 -25
  96. package/dist/esm/tools/ToolNode.mjs.map +1 -1
  97. package/dist/esm/tools/handlers.mjs +1 -1
  98. package/dist/esm/tools/search/tool.mjs +1 -1
  99. package/dist/esm/tools/subagent/SubagentExecutor.mjs +1 -1
  100. package/dist/esm/utils/index.mjs +2 -2
  101. package/dist/types/graphs/Graph.d.ts +19 -0
  102. package/dist/types/hitl/askUserQuestion.d.ts +11 -1
  103. package/dist/types/langfuse.d.ts +16 -8
  104. package/dist/types/langfuseConfig.d.ts +6 -0
  105. package/dist/types/langfuseRuntimeContext.d.ts +27 -1
  106. package/dist/types/langfuseRuntimeScope.d.ts +17 -2
  107. package/dist/types/langfuseSpanRegistry.d.ts +17 -0
  108. package/dist/types/langfuseTraceShaping.d.ts +2 -1
  109. package/dist/types/llm/anthropic/utils/message_inputs.d.ts +1 -0
  110. package/dist/types/messages/format.d.ts +9 -8
  111. package/dist/types/prompts/activityLabel.d.ts +8 -1
  112. package/dist/types/run.d.ts +1 -1
  113. package/dist/types/session/types.d.ts +1 -0
  114. package/dist/types/tools/ToolNode.d.ts +7 -1
  115. package/dist/types/types/activityLabel.d.ts +8 -0
  116. package/dist/types/types/hitl.d.ts +8 -0
  117. package/dist/types/types/stream.d.ts +19 -0
  118. package/dist/types/types/tools.d.ts +30 -0
  119. package/package.json +7 -4
  120. package/src/__tests__/stream.eagerArgsDivergence.test.ts +753 -0
  121. package/src/graphs/Graph.ts +69 -20
  122. package/src/graphs/MultiAgentGraph.ts +74 -6
  123. package/src/graphs/__tests__/composition.smoke.test.ts +4 -0
  124. package/src/hitl/askUserQuestion.ts +14 -1
  125. package/src/instrumentation.ts +35 -77
  126. package/src/langfuse.ts +320 -43
  127. package/src/langfuseConfig.ts +24 -0
  128. package/src/langfuseRuntimeContext.ts +43 -1
  129. package/src/langfuseRuntimeScope.ts +94 -21
  130. package/src/langfuseSpanRegistry.ts +131 -0
  131. package/src/langfuseTraceShaping.ts +194 -7
  132. package/src/llm/anthropic/utils/message_inputs.ts +70 -19
  133. package/src/llm/anthropic/utils/streaming-tool-input.test.ts +186 -11
  134. package/src/llm/bedrock/utils/message_inputs.test.ts +120 -4
  135. package/src/llm/bedrock/utils/message_inputs.ts +32 -7
  136. package/src/messages/format.ts +222 -50
  137. package/src/messages/formatAgentMessages.test.ts +308 -6
  138. package/src/messages/injected.test.ts +18 -1
  139. package/src/messages/injected.ts +8 -1
  140. package/src/messages/prune.ts +12 -1
  141. package/src/prompts/activityLabel.ts +67 -2
  142. package/src/run.ts +86 -46
  143. package/src/scripts/activity-labels/captured.json +56 -0
  144. package/src/scripts/activity-labels/checks.cjs +205 -0
  145. package/src/scripts/activity-labels/corpus.cjs +473 -0
  146. package/src/scripts/activity-labels/report.cjs +203 -0
  147. package/src/scripts/activity-labels/rescore.cjs +102 -0
  148. package/src/scripts/activity-labels/run.ts +705 -0
  149. package/src/scripts/activity-labels/variants.ts +71 -0
  150. package/src/session/messageSerialization.ts +12 -1
  151. package/src/session/types.ts +1 -0
  152. package/src/specs/activity-label-prompt.test.ts +109 -0
  153. package/src/specs/agent-handoffs.test.ts +306 -0
  154. package/src/specs/langfuse-callbacks.test.ts +456 -0
  155. package/src/specs/langfuse-routing.integration.test.ts +138 -1
  156. package/src/specs/langfuse-span-registry.test.ts +70 -0
  157. package/src/specs/langfuse-trace-shaping.test.ts +294 -0
  158. package/src/specs/prune.test.ts +38 -1
  159. package/src/stream.ts +70 -6
  160. package/src/summarization/__tests__/node.test.ts +188 -0
  161. package/src/summarization/node.ts +72 -0
  162. package/src/tools/ToolNode.ts +400 -9
  163. package/src/tools/__tests__/ToolNode.invalidToolCalls.test.ts +757 -0
  164. package/src/tools/__tests__/hitl.test.ts +58 -0
  165. package/src/types/activityLabel.ts +8 -0
  166. package/src/types/hitl.ts +8 -0
  167. package/src/types/stream.ts +20 -0
  168. package/src/types/tools.ts +35 -1
@@ -11,6 +11,7 @@ import {
11
11
  convertMessagesToResponsesInput,
12
12
  convertResponsesMessageToAIMessage,
13
13
  } from '@langchain/openai';
14
+ import type { BaseMessage } from '@langchain/core/messages';
14
15
  import type { MessageContentComplex, TPayload } from '@/types';
15
16
  import {
16
17
  convertMessagesToContent,
@@ -4782,6 +4783,119 @@ describe('formatAgentMessages', () => {
4782
4783
  });
4783
4784
 
4784
4785
  describe('summary boundary token count adjustment', () => {
4786
+ /** Atomic media costs a fixed provider price the character heuristic cannot
4787
+ * see, so scaling by the measurable siblings alone erases it. Both shapes
4788
+ * collapsed a four-figure count to 1 before this guard. */
4789
+ it.each([
4790
+ [
4791
+ 'text before the summary',
4792
+ { type: ContentTypes.TEXT, text: 'hello there' },
4793
+ ],
4794
+ [
4795
+ 'a tool call before the summary',
4796
+ {
4797
+ type: ContentTypes.TOOL_CALL,
4798
+ tool_call: {
4799
+ id: 'tc1',
4800
+ name: 'search',
4801
+ args: '{"q":"x"}',
4802
+ output: 'result text',
4803
+ },
4804
+ },
4805
+ ],
4806
+ ])(
4807
+ 'skips the positional discount when retained media is unmeasurable, with %s',
4808
+ (_label, leading) => {
4809
+ const payload: TPayload = [
4810
+ {
4811
+ role: 'assistant',
4812
+ content: [
4813
+ leading as MessageContentComplex,
4814
+ {
4815
+ type: ContentTypes.SUMMARY,
4816
+ text: 'S'.repeat(400),
4817
+ tokenCount: 100,
4818
+ },
4819
+ {
4820
+ type: 'image_url',
4821
+ image_url: { url: 'data:image/png;base64,x' },
4822
+ },
4823
+ ],
4824
+ },
4825
+ ];
4826
+
4827
+ const result = formatAgentMessages(payload, { 0: 1200 });
4828
+
4829
+ expect(result.indexTokenCountMap?.[0]).toBe(1200);
4830
+ expect(result.boundaryTokenAdjustment).toBeUndefined();
4831
+ }
4832
+ );
4833
+
4834
+ /** The media sits a level down, inside `tool_call.output`, where serializing
4835
+ * gives it a nonzero length while the token counter charges its fixed media
4836
+ * cost. Eligibility is decided by part type, so nesting depth is irrelevant. */
4837
+ it('skips the positional discount when retained tool output carries media', () => {
4838
+ const payload: TPayload = [
4839
+ {
4840
+ role: 'assistant',
4841
+ content: [
4842
+ { type: ContentTypes.TEXT, text: 'a'.repeat(4000) },
4843
+ {
4844
+ type: ContentTypes.SUMMARY,
4845
+ text: 'S'.repeat(400),
4846
+ tokenCount: 100,
4847
+ },
4848
+ {
4849
+ type: ContentTypes.TOOL_CALL,
4850
+ tool_call: {
4851
+ id: 'tc1',
4852
+ name: 'render',
4853
+ args: '{}',
4854
+ output: [
4855
+ {
4856
+ type: 'image_url',
4857
+ image_url: { url: 'data:image/png;base64,y' },
4858
+ },
4859
+ ],
4860
+ },
4861
+ },
4862
+ ],
4863
+ },
4864
+ ];
4865
+
4866
+ const result = formatAgentMessages(payload, { 0: 4000 });
4867
+
4868
+ const emitted = Object.values(result.indexTokenCountMap ?? {}).reduce(
4869
+ (sum, value) => sum + value,
4870
+ 0
4871
+ );
4872
+ expect(emitted).toBe(4000);
4873
+ expect(result.boundaryTokenAdjustment).toBeUndefined();
4874
+ });
4875
+
4876
+ it('still proportions when every retained part is measurable', () => {
4877
+ const payload: TPayload = [
4878
+ {
4879
+ role: 'assistant',
4880
+ content: [
4881
+ { type: ContentTypes.TEXT, text: 'a'.repeat(400) },
4882
+ {
4883
+ type: ContentTypes.SUMMARY,
4884
+ text: 'S'.repeat(100),
4885
+ tokenCount: 20,
4886
+ },
4887
+ { type: ContentTypes.TEXT, text: 'b'.repeat(100) },
4888
+ ],
4889
+ },
4890
+ ];
4891
+
4892
+ const result = formatAgentMessages(payload, { 0: 600 });
4893
+
4894
+ expect(result.boundaryTokenAdjustment?.original).toBe(600);
4895
+ expect(result.indexTokenCountMap?.[0]).toBeLessThan(600);
4896
+ expect(result.indexTokenCountMap?.[0]).toBeGreaterThan(0);
4897
+ });
4898
+
4785
4899
  it('should proportion token count when thinking block is sliced off by boundary', () => {
4786
4900
  const thinkingText = 'x'.repeat(1000);
4787
4901
  const payload: TPayload = [
@@ -4815,7 +4929,10 @@ describe('formatAgentMessages', () => {
4815
4929
  expect(result.indexTokenCountMap?.[1]).toBe(8);
4816
4930
  });
4817
4931
 
4818
- it('should proportion token count when thinking + tool_use are sliced off', () => {
4932
+ /** Reframed: a tool call anywhere in the entry now cancels the ratio, since
4933
+ * telling a text-bearing tool payload from a media-bearing one requires
4934
+ * recursing into arbitrary nested output. The entry keeps its count. */
4935
+ it('should not proportion when a tool_use part is present', () => {
4819
4936
  const thinkingText = 'a'.repeat(800);
4820
4937
  const toolInput = JSON.stringify({ data: 'b'.repeat(400) });
4821
4938
  const payload: TPayload = [
@@ -4851,8 +4968,8 @@ describe('formatAgentMessages', () => {
4851
4968
  result.indexTokenCountMap || {}
4852
4969
  ).reduce((sum, v) => sum + v, 0);
4853
4970
 
4854
- expect(totalOutputTokens).toBeLessThan(200);
4855
- expect(totalOutputTokens).toBeGreaterThan(0);
4971
+ expect(totalOutputTokens).toBe(2000);
4972
+ expect(result.boundaryTokenAdjustment).toBeUndefined();
4856
4973
  });
4857
4974
 
4858
4975
  it('should roughly halve token count when content is evenly split around boundary', () => {
@@ -4908,7 +5025,11 @@ describe('formatAgentMessages', () => {
4908
5025
  expect(result.indexTokenCountMap?.[1]).toBe(10);
4909
5026
  });
4910
5027
 
4911
- it('should account for tool_use input size in the char-length ratio', () => {
5028
+ /** Previously the removed `tool_use` input was counted into the denominator.
5029
+ * A base64 payload there serializes to a huge length while the counter
5030
+ * charges a fixed estimate, so the ratio dragged retained text below its
5031
+ * real cost. The discount is cancelled instead. */
5032
+ it('should not use tool_use input size in the char-length ratio', () => {
4912
5033
  const hugeInput = JSON.stringify({ payload: 'z'.repeat(5000) });
4913
5034
  const payload: TPayload = [
4914
5035
  {
@@ -4934,8 +5055,8 @@ describe('formatAgentMessages', () => {
4934
5055
  expect(result.summary).toBeDefined();
4935
5056
 
4936
5057
  const adjustedTokens = result.indexTokenCountMap?.[0] ?? 0;
4937
- expect(adjustedTokens).toBeLessThan(100);
4938
- expect(adjustedTokens).toBeGreaterThan(0);
5058
+ expect(adjustedTokens).toBe(3000);
5059
+ expect(result.boundaryTokenAdjustment).toBeUndefined();
4939
5060
  });
4940
5061
 
4941
5062
  it('should handle multiple content parts after the boundary', () => {
@@ -4997,6 +5118,187 @@ describe('formatAgentMessages', () => {
4997
5118
  });
4998
5119
  });
4999
5120
 
5121
+ describe('summary coverage boundary', () => {
5122
+ const buildSummaryPart = (coverage?: {
5123
+ retainedFromMessageId: string;
5124
+ }) => ({
5125
+ type: ContentTypes.SUMMARY,
5126
+ content: [
5127
+ { type: ContentTypes.TEXT, text: 'Summary of the earliest turns' },
5128
+ ],
5129
+ tokenCount: 12,
5130
+ ...(coverage != null ? { coverage } : {}),
5131
+ });
5132
+
5133
+ /** Mirrors a compaction with `retainRecent.turns: 1`: m1/m2 were refined
5134
+ * into the summary, m3/m4 are the retained tail, and the block itself is
5135
+ * persisted on the assistant message that came after all of them. */
5136
+ const compactedPayload = (coverage?: {
5137
+ retainedFromMessageId: string;
5138
+ }): TPayload => [
5139
+ { messageId: 'm1', role: 'user', content: 'Covered question' },
5140
+ { messageId: 'm2', role: 'assistant', content: 'Covered answer' },
5141
+ { messageId: 'm3', role: 'user', content: 'Retained question' },
5142
+ { messageId: 'm4', role: 'assistant', content: 'Retained answer' },
5143
+ {
5144
+ messageId: 'm5',
5145
+ role: 'assistant',
5146
+ content: [
5147
+ buildSummaryPart(coverage),
5148
+ { type: ContentTypes.TEXT, text: 'Post-compaction reply' },
5149
+ ],
5150
+ },
5151
+ ];
5152
+
5153
+ const textOf = (message: BaseMessage): string => {
5154
+ const { content } = message;
5155
+ if (typeof content === 'string') {
5156
+ return content;
5157
+ }
5158
+ return (content as MessageContentComplex[])
5159
+ .map((part) => ('text' in part ? (part as { text: string }).text : ''))
5160
+ .join('');
5161
+ };
5162
+
5163
+ it('preserves the retained tail that the summary never covered', () => {
5164
+ const result = formatAgentMessages(
5165
+ compactedPayload({ retainedFromMessageId: 'm3' })
5166
+ );
5167
+
5168
+ expect(result.messages.map(textOf)).toEqual([
5169
+ 'Retained question',
5170
+ 'Retained answer',
5171
+ 'Post-compaction reply',
5172
+ ]);
5173
+ expect(result.summary!.text).toBe('Summary of the earliest turns');
5174
+ expect(result.summary!.tokenCount).toBe(12);
5175
+ });
5176
+
5177
+ it('retains the anchor message itself, dropping only what precedes it', () => {
5178
+ const result = formatAgentMessages(
5179
+ compactedPayload({ retainedFromMessageId: 'm4' })
5180
+ );
5181
+
5182
+ expect(result.messages.map(textOf)).toEqual([
5183
+ 'Retained answer',
5184
+ 'Post-compaction reply',
5185
+ ]);
5186
+ });
5187
+
5188
+ /** Coverage mode leaves the block's entry at its full count on purpose. The
5189
+ * summary's cost in the reader's token units is not obtainable here — no
5190
+ * tokenizer reaches this function, and a figure recorded at write time is
5191
+ * in the writing run's units. Over-counting prunes early; under-counting
5192
+ * would risk an over-context request. */
5193
+ it('does not discount the entry carrying the summary block', () => {
5194
+ const payload: TPayload = [
5195
+ { messageId: 'm1', role: 'user', content: 'Covered question' },
5196
+ { messageId: 'm2', role: 'user', content: 'Retained question' },
5197
+ {
5198
+ messageId: 'm3',
5199
+ role: 'assistant',
5200
+ content: [
5201
+ {
5202
+ type: ContentTypes.SUMMARY,
5203
+ content: [{ type: ContentTypes.TEXT, text: 'S'.repeat(500) }],
5204
+ tokenCount: 120,
5205
+ coverage: { retainedFromMessageId: 'm2' },
5206
+ },
5207
+ { type: ContentTypes.TEXT, text: 'Reply' },
5208
+ ],
5209
+ },
5210
+ ];
5211
+
5212
+ const result = formatAgentMessages(payload, { 0: 5, 1: 6, 2: 1000 });
5213
+
5214
+ expect(result.indexTokenCountMap?.[1]).toBe(1000);
5215
+ expect(result.boundaryTokenAdjustment).toBeUndefined();
5216
+ });
5217
+
5218
+ it('keeps token counts for the retained tail and drops covered entries', () => {
5219
+ const result = formatAgentMessages(
5220
+ compactedPayload({ retainedFromMessageId: 'm3' }),
5221
+ { 0: 5, 1: 6, 2: 7, 3: 8, 4: 40 }
5222
+ );
5223
+
5224
+ expect(result.indexTokenCountMap?.[0]).toBe(7);
5225
+ expect(result.indexTokenCountMap?.[1]).toBe(8);
5226
+ expect(Object.keys(result.indexTokenCountMap ?? {})).toHaveLength(3);
5227
+ });
5228
+
5229
+ it('leaves entries without summary parts untouched', () => {
5230
+ const result = formatAgentMessages(
5231
+ compactedPayload({ retainedFromMessageId: 'm3' }),
5232
+ { 0: 5, 1: 6, 2: 7, 3: 8, 4: 40 }
5233
+ );
5234
+
5235
+ expect(result.indexTokenCountMap?.[0]).toBe(7);
5236
+ expect(result.indexTokenCountMap?.[1]).toBe(8);
5237
+ });
5238
+
5239
+ it('falls back to positional trimming for legacy blocks without coverage', () => {
5240
+ const result = formatAgentMessages(compactedPayload());
5241
+
5242
+ expect(result.messages.map(textOf)).toEqual(['Post-compaction reply']);
5243
+ expect(result.summary!.text).toBe('Summary of the earliest turns');
5244
+ });
5245
+
5246
+ it('falls back to positional trimming when coverage cannot be resolved', () => {
5247
+ const result = formatAgentMessages(
5248
+ compactedPayload({ retainedFromMessageId: 'pruned-from-payload' })
5249
+ );
5250
+
5251
+ expect(result.messages.map(textOf)).toEqual(['Post-compaction reply']);
5252
+ });
5253
+
5254
+ it('ignores an anchor pointing past its own block', () => {
5255
+ const result = formatAgentMessages([
5256
+ ...compactedPayload({ retainedFromMessageId: 'm6' }),
5257
+ { messageId: 'm6', role: 'user', content: 'Later question' },
5258
+ ]);
5259
+
5260
+ expect(result.messages.map(textOf)).toEqual([
5261
+ 'Post-compaction reply',
5262
+ 'Later question',
5263
+ ]);
5264
+ });
5265
+
5266
+ it('applies last-summary-wins across mixed coverage and legacy blocks', () => {
5267
+ const payload: TPayload = [
5268
+ { messageId: 'm1', role: 'user', content: 'Covered question' },
5269
+ {
5270
+ messageId: 'm2',
5271
+ role: 'assistant',
5272
+ content: [
5273
+ {
5274
+ type: ContentTypes.SUMMARY,
5275
+ text: 'Older summary',
5276
+ tokenCount: 3,
5277
+ },
5278
+ { type: ContentTypes.TEXT, text: 'Older tail' },
5279
+ ],
5280
+ },
5281
+ { messageId: 'm3', role: 'user', content: 'Retained question' },
5282
+ {
5283
+ messageId: 'm4',
5284
+ role: 'assistant',
5285
+ content: [
5286
+ buildSummaryPart({ retainedFromMessageId: 'm3' }),
5287
+ { type: ContentTypes.TEXT, text: 'Newest reply' },
5288
+ ],
5289
+ },
5290
+ ];
5291
+
5292
+ const result = formatAgentMessages(payload);
5293
+
5294
+ expect(result.messages.map(textOf)).toEqual([
5295
+ 'Retained question',
5296
+ 'Newest reply',
5297
+ ]);
5298
+ expect(result.summary!.text).toBe('Summary of the earliest turns');
5299
+ });
5300
+ });
5301
+
5000
5302
  describe('cross-run summary token accounting', () => {
5001
5303
  it('should conserve tokens: summary boundary excludes pre-boundary messages from the map', () => {
5002
5304
  const payload: TPayload = [
@@ -34,7 +34,7 @@ describe('convertInjectedMessages', () => {
34
34
 
35
35
  it('carries isMeta, source and skillName only when set', () => {
36
36
  const [bare] = convertInjectedMessages([{ role: 'user', content: 'x' }]);
37
- expect(bare.additional_kwargs).toEqual({ role: 'user' });
37
+ expect(bare.additional_kwargs).toEqual({ role: 'user', injected: true });
38
38
 
39
39
  const [full] = convertInjectedMessages([
40
40
  {
@@ -47,12 +47,29 @@ describe('convertInjectedMessages', () => {
47
47
  ]);
48
48
  expect(full.additional_kwargs).toEqual({
49
49
  role: 'user',
50
+ injected: true,
50
51
  isMeta: true,
51
52
  source: 'steer',
52
53
  skillName: 'writing',
53
54
  });
54
55
  });
55
56
 
57
+ /** Both marker fields are optional on `InjectedMessage`, so consumers that
58
+ * must tell in-run context from payload-replayed messages — compaction
59
+ * coverage anchors — cannot rely on them. `injected` is unconditional. */
60
+ it('always records injected provenance, whatever the caller supplied', () => {
61
+ const converted = convertInjectedMessages([
62
+ { role: 'user', content: 'bare' },
63
+ { role: 'system', content: 'hook output', source: 'hook' },
64
+ { role: 'user', content: 'injected steer', source: 'steer' },
65
+ ]);
66
+
67
+ expect(converted).toHaveLength(3);
68
+ for (const message of converted) {
69
+ expect(message.additional_kwargs.injected).toBe(true);
70
+ }
71
+ });
72
+
56
73
  it('passes multimodal content through as a content array', () => {
57
74
  const [converted] = convertInjectedMessages([
58
75
  {
@@ -2,8 +2,8 @@
2
2
  import { HumanMessage } from '@langchain/core/messages';
3
3
  import type { BaseMessage } from '@langchain/core/messages';
4
4
  import type { InjectedMessage } from '@/types/tools';
5
- import { ContentTypes } from '@/common';
6
5
  import { toLangChainContent } from './langchain';
6
+ import { ContentTypes } from '@/common';
7
7
 
8
8
  /**
9
9
  * Converts `InjectedMessage` instances to LangChain `HumanMessage` objects.
@@ -56,8 +56,15 @@ export function convertInjectedMessages(
56
56
  if (isEmptyInjectedContent(msg.content)) {
57
57
  continue;
58
58
  }
59
+ /** Provenance, recorded here because this is the only place that knows it.
60
+ * `isMeta` and `source` are both optional on `InjectedMessage`, so a bare
61
+ * entry is otherwise indistinguishable from a message replayed out of the
62
+ * payload — and downstream consumers such as compaction coverage need to
63
+ * know that this message has no persisted source ID to name. Kept separate
64
+ * from `isMeta`, which carries UI and cache meaning of its own. */
59
65
  const additional_kwargs: Record<string, unknown> = {
60
66
  role: msg.role,
67
+ injected: true,
61
68
  };
62
69
  if (msg.isMeta != null) additional_kwargs.isMeta = msg.isMeta;
63
70
  if (msg.source != null) additional_kwargs.source = msg.source;
@@ -1403,7 +1403,18 @@ function createBoundedTruncationValue(
1403
1403
  _originalChars: originalChars,
1404
1404
  };
1405
1405
  if (JSON.stringify(emptyEnvelope).length > normalizedMaxChars) {
1406
- return null;
1406
+ /**
1407
+ * Even the empty envelope overflows the cap, so no preview survives —
1408
+ * but the result must still be a JSON OBJECT, never `null`. This value
1409
+ * replaces a `tool_use.input` / tool-call `args` on messages that are
1410
+ * mutated IN PLACE into graph state (`preFlightTruncateToolCallInputs`),
1411
+ * and Anthropic rejects a replayed non-object input with a 400
1412
+ * (`tool_use.input: Input should be an object`). Observed live: a tight
1413
+ * summarization budget shrank the cap below the envelope, nulled a
1414
+ * retained calculator call's input and args, and the next model call
1415
+ * failed on replay.
1416
+ */
1417
+ return {};
1407
1418
  }
1408
1419
 
1409
1420
  let low = 0;
@@ -33,6 +33,26 @@ export function truncateForLabel(value: string, maxLength: number): string {
33
33
  return value.slice(0, Math.max(0, maxLength - 1)) + '…';
34
34
  }
35
35
 
36
+ /**
37
+ * Reduces a committed label to bounded single-line data.
38
+ *
39
+ * Sections in this prompt are delimited by blank lines, so a label carrying
40
+ * embedded newlines could otherwise forge an apparent entries section or
41
+ * `Header:` cue. Unlike every other input here, previous labels re-enter
42
+ * the prompt on EVERY later batch, so one malformed result — plain model
43
+ * noncompliance, or injection surfacing through a tool result — would
44
+ * persistently steer unrelated later labels rather than affecting one. The
45
+ * clip bounds the same way `lastAssistantText` and reasoning excerpts are
46
+ * bounded: oversized headers must not inflate later requests past the fast
47
+ * model's window and starve the run of labels entirely.
48
+ */
49
+ function sanitizePreviousLabel(label: string): string {
50
+ return truncateForLabel(
51
+ label.replace(/\s+/g, ' ').trim(),
52
+ PREVIOUS_LABEL_LIMIT
53
+ );
54
+ }
55
+
36
56
  const ABORT_SERIALIZATION = Symbol('abort-label-serialization');
37
57
 
38
58
  /**
@@ -76,6 +96,11 @@ function serializeForLabel(value: unknown, limit: number): string {
76
96
 
77
97
  const INPUT_CONTEXT_LIMIT = 200;
78
98
  const MAX_THINKING_EXCERPTS = 4;
99
+ const MAX_PREVIOUS_LABELS = 3;
100
+ /** Per-label bound. A header is 5-9 words; anything past this is
101
+ * noncompliance or payload, and previous labels are the one input that
102
+ * RE-ENTERS the prompt on every later batch of the run. */
103
+ const PREVIOUS_LABEL_LIMIT = 200;
79
104
  /** A label is 5-9 words; no batch needs more than this many entries to
80
105
  * produce one, and the cap keeps a 200-call programmatic batch from
81
106
  * building an enormous prompt out of per-field-bounded pieces. */
@@ -86,6 +111,13 @@ export type BuildActivityLabelPromptParams = {
86
111
  charLimit: number;
87
112
  thinkingExcerpts?: string[];
88
113
  lastAssistantText?: string;
114
+ /**
115
+ * Headers already committed for earlier batches in this run, in run order
116
+ * with the most recent last. Rendered ahead of the block context so the
117
+ * label continues the run's story instead of restating a line the user is
118
+ * already reading. Capped at {@link MAX_PREVIOUS_LABELS}.
119
+ */
120
+ previousLabels?: string[];
89
121
  /**
90
122
  * Resolved tool-output tracing policy. The label prompt becomes Langfuse
91
123
  * generation input, so outputs/errors excluded from tracing (global
@@ -104,6 +136,7 @@ export function buildActivityLabelPrompt({
104
136
  charLimit,
105
137
  thinkingExcerpts,
106
138
  lastAssistantText,
139
+ previousLabels,
107
140
  redaction,
108
141
  }: BuildActivityLabelPromptParams): string {
109
142
  const clip = truncateForLabel;
@@ -116,6 +149,28 @@ export function buildActivityLabelPrompt({
116
149
  redaction != null &&
117
150
  (redaction.enabled === false || redaction.redactedToolNames.size > 0);
118
151
  const sections: string[] = [];
152
+ /** Previous labels are free-form model prose too, and per-agent overlays
153
+ * mean an earlier header may have been generated under ANOTHER agent's
154
+ * weaker policy — so they share the excerpts' wholesale drop rather than
155
+ * letting a handoff leak a looser agent's phrasing into this trace. */
156
+ if (
157
+ !excerptsRedacted &&
158
+ previousLabels != null &&
159
+ previousLabels.length > 0
160
+ ) {
161
+ const recent = previousLabels
162
+ .slice(-MAX_PREVIOUS_LABELS)
163
+ .map(sanitizePreviousLabel)
164
+ /** A label that sanitizes to nothing carries no story to continue;
165
+ * rendering it would leave a bare bullet implying a missing header. */
166
+ .filter((label) => label.length > 0);
167
+ if (recent.length > 0) {
168
+ sections.push(
169
+ 'Previous headers in this run (most recent last):\n' +
170
+ recent.map((label) => `- ${label}`).join('\n')
171
+ );
172
+ }
173
+ }
119
174
  /** Intent text is free-form assistant prose that can quote a redacted
120
175
  * tool result just as reasoning can, so it shares the excerpts' fate. */
121
176
  if (
@@ -144,7 +199,14 @@ export function buildActivityLabelPrompt({
144
199
  const shown = entries.slice(0, MAX_PROMPT_ENTRIES);
145
200
  const omitted = entries.length - shown.length;
146
201
  sections.push(
147
- 'Tool calls:\n' +
202
+ /** Frames the list as reference material, not the thing to
203
+ * transcribe. Ported from LibreChat's fallback builder (its
204
+ * runtime.ts documents that without this the model "hands back a
205
+ * transcription" of the list) after the eval harness measured it
206
+ * across three independent sweeps: fewer template-redundancy and
207
+ * length violations than a bare `Tool calls:` heading, with no
208
+ * per-case regressions (agents #360). */
209
+ 'What it called, and what came back (do not restate these):\n' +
148
210
  shown
149
211
  .map((entry) => {
150
212
  const input = clip(
@@ -172,6 +234,9 @@ export function buildActivityLabelPrompt({
172
234
  : '')
173
235
  );
174
236
  }
175
- sections.push('Label:');
237
+ /** The fallback builder's terminal cue, measured alongside the heading
238
+ * (same sweeps). The default system prompt already describes the
239
+ * output as "the header of a collapsed activity group". */
240
+ sections.push('Header:');
176
241
  return sections.join('\n\n');
177
242
  }