@librechat/agents 3.7.15 → 3.7.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/dist/cjs/agents/AgentContext.cjs +91 -0
  2. package/dist/cjs/agents/AgentContext.cjs.map +1 -1
  3. package/dist/cjs/agents/projection.cjs +2 -1
  4. package/dist/cjs/agents/projection.cjs.map +1 -1
  5. package/dist/cjs/graphs/Graph.cjs +47 -6
  6. package/dist/cjs/graphs/Graph.cjs.map +1 -1
  7. package/dist/cjs/graphs/MultiAgentGraph.cjs +20 -7
  8. package/dist/cjs/graphs/MultiAgentGraph.cjs.map +1 -1
  9. package/dist/cjs/llm/anthropic/utils/message_inputs.cjs +1 -1
  10. package/dist/cjs/llm/openai/utils/index.cjs +1 -1
  11. package/dist/cjs/llm/openai/utils/index.cjs.map +1 -1
  12. package/dist/cjs/llm/preempt.cjs +1 -10
  13. package/dist/cjs/llm/preempt.cjs.map +1 -1
  14. package/dist/cjs/main.cjs +23 -3
  15. package/dist/cjs/messages/contextPruning.cjs +1 -1
  16. package/dist/cjs/messages/contextPruning.cjs.map +1 -1
  17. package/dist/cjs/messages/core.cjs +29 -1
  18. package/dist/cjs/messages/core.cjs.map +1 -1
  19. package/dist/cjs/messages/fading.cjs +122 -0
  20. package/dist/cjs/messages/fading.cjs.map +1 -0
  21. package/dist/cjs/messages/format.cjs +1 -1
  22. package/dist/cjs/messages/index.cjs +1 -0
  23. package/dist/cjs/messages/prune.cjs +364 -236
  24. package/dist/cjs/messages/prune.cjs.map +1 -1
  25. package/dist/cjs/run.cjs +27 -0
  26. package/dist/cjs/run.cjs.map +1 -1
  27. package/dist/cjs/session/AgentSession.cjs +213 -8
  28. package/dist/cjs/session/AgentSession.cjs.map +1 -1
  29. package/dist/cjs/session/JsonlSessionStore.cjs +41 -0
  30. package/dist/cjs/session/JsonlSessionStore.cjs.map +1 -1
  31. package/dist/cjs/session/index.cjs +1 -1
  32. package/dist/cjs/stream.cjs +1 -1
  33. package/dist/cjs/summarization/node.cjs +1 -1
  34. package/dist/cjs/tools/ToolNode.cjs +1 -1
  35. package/dist/cjs/tools/subagent/SubagentExecutor.cjs +2 -0
  36. package/dist/cjs/tools/subagent/SubagentExecutor.cjs.map +1 -1
  37. package/dist/cjs/tools/subagent/SubagentReplay.cjs +2 -1
  38. package/dist/cjs/tools/subagent/SubagentReplay.cjs.map +1 -1
  39. package/dist/cjs/utils/errors.cjs +3 -1
  40. package/dist/cjs/utils/errors.cjs.map +1 -1
  41. package/dist/cjs/utils/index.cjs +1 -1
  42. package/dist/cjs/utils/truncation.cjs +9 -0
  43. package/dist/cjs/utils/truncation.cjs.map +1 -1
  44. package/dist/esm/agents/AgentContext.mjs +92 -1
  45. package/dist/esm/agents/AgentContext.mjs.map +1 -1
  46. package/dist/esm/agents/projection.mjs +2 -1
  47. package/dist/esm/agents/projection.mjs.map +1 -1
  48. package/dist/esm/graphs/Graph.mjs +47 -6
  49. package/dist/esm/graphs/Graph.mjs.map +1 -1
  50. package/dist/esm/graphs/MultiAgentGraph.mjs +20 -7
  51. package/dist/esm/graphs/MultiAgentGraph.mjs.map +1 -1
  52. package/dist/esm/llm/anthropic/utils/message_inputs.mjs +1 -1
  53. package/dist/esm/llm/openai/utils/index.mjs +2 -2
  54. package/dist/esm/llm/openai/utils/index.mjs.map +1 -1
  55. package/dist/esm/llm/preempt.mjs +1 -10
  56. package/dist/esm/llm/preempt.mjs.map +1 -1
  57. package/dist/esm/main.mjs +6 -5
  58. package/dist/esm/messages/contextPruning.mjs +1 -1
  59. package/dist/esm/messages/contextPruning.mjs.map +1 -1
  60. package/dist/esm/messages/core.mjs +29 -2
  61. package/dist/esm/messages/core.mjs.map +1 -1
  62. package/dist/esm/messages/fading.mjs +108 -0
  63. package/dist/esm/messages/fading.mjs.map +1 -0
  64. package/dist/esm/messages/format.mjs +1 -1
  65. package/dist/esm/messages/index.mjs +1 -0
  66. package/dist/esm/messages/prune.mjs +363 -235
  67. package/dist/esm/messages/prune.mjs.map +1 -1
  68. package/dist/esm/run.mjs +27 -0
  69. package/dist/esm/run.mjs.map +1 -1
  70. package/dist/esm/session/AgentSession.mjs +213 -8
  71. package/dist/esm/session/AgentSession.mjs.map +1 -1
  72. package/dist/esm/session/JsonlSessionStore.mjs +41 -0
  73. package/dist/esm/session/JsonlSessionStore.mjs.map +1 -1
  74. package/dist/esm/session/index.mjs +1 -1
  75. package/dist/esm/stream.mjs +1 -1
  76. package/dist/esm/summarization/node.mjs +1 -1
  77. package/dist/esm/tools/ToolNode.mjs +1 -1
  78. package/dist/esm/tools/subagent/SubagentExecutor.mjs +2 -0
  79. package/dist/esm/tools/subagent/SubagentExecutor.mjs.map +1 -1
  80. package/dist/esm/tools/subagent/SubagentReplay.mjs +2 -1
  81. package/dist/esm/tools/subagent/SubagentReplay.mjs.map +1 -1
  82. package/dist/esm/utils/errors.mjs +3 -1
  83. package/dist/esm/utils/errors.mjs.map +1 -1
  84. package/dist/esm/utils/index.mjs +1 -1
  85. package/dist/esm/utils/truncation.mjs +7 -1
  86. package/dist/esm/utils/truncation.mjs.map +1 -1
  87. package/dist/types/agents/AgentContext.d.ts +32 -1
  88. package/dist/types/agents/projection.d.ts +3 -1
  89. package/dist/types/graphs/Graph.d.ts +14 -1
  90. package/dist/types/graphs/MultiAgentGraph.d.ts +7 -0
  91. package/dist/types/messages/contextPruning.d.ts +2 -0
  92. package/dist/types/messages/core.d.ts +2 -0
  93. package/dist/types/messages/fading.d.ts +81 -0
  94. package/dist/types/messages/index.d.ts +1 -0
  95. package/dist/types/messages/prune.d.ts +71 -24
  96. package/dist/types/run.d.ts +16 -0
  97. package/dist/types/session/AgentSession.d.ts +15 -0
  98. package/dist/types/session/JsonlSessionStore.d.ts +4 -1
  99. package/dist/types/session/types.d.ts +9 -0
  100. package/dist/types/tools/subagent/SubagentReplay.d.ts +3 -1
  101. package/dist/types/types/graph.d.ts +23 -0
  102. package/dist/types/types/run.d.ts +10 -0
  103. package/dist/types/utils/truncation.d.ts +9 -0
  104. package/package.json +1 -1
  105. package/src/agents/AgentContext.ts +175 -1
  106. package/src/agents/projection.ts +4 -0
  107. package/src/graphs/Graph.ts +97 -11
  108. package/src/graphs/MultiAgentGraph.ts +54 -6
  109. package/src/llm/openai/utils/index.ts +3 -1
  110. package/src/llm/preempt.ts +4 -17
  111. package/src/messages/contextPruning.ts +7 -1
  112. package/src/messages/core.ts +52 -0
  113. package/src/messages/fading.ts +301 -0
  114. package/src/messages/index.ts +1 -0
  115. package/src/messages/prune.ts +719 -507
  116. package/src/run.ts +52 -0
  117. package/src/session/AgentSession.ts +397 -8
  118. package/src/session/JsonlSessionStore.ts +71 -0
  119. package/src/session/types.ts +10 -0
  120. package/src/tools/subagent/SubagentExecutor.ts +6 -0
  121. package/src/tools/subagent/SubagentReplay.ts +15 -2
  122. package/src/types/graph.ts +25 -0
  123. package/src/types/run.ts +10 -0
  124. package/src/utils/errors.ts +17 -1
  125. package/src/utils/truncation.ts +25 -0
@@ -1,5 +1,6 @@
1
1
  import { BaseMessage, UsageMetadata } from '@langchain/core/messages';
2
- import type { ContextPruningConfig } from '@/types/graph';
2
+ import type { ContextPruningConfig, FadingTier } from '@/types/graph';
3
+ import type { FadingCaps } from './fading';
3
4
  import type { TokenCounter } from '@/types/run';
4
5
  import type { ProviderName } from '@/types';
5
6
  import { ContentTypes } from '@/common';
@@ -65,10 +66,29 @@ export type PruneMessagesFactoryParams = {
65
66
  * of waiting for the first provider response. Ignored when <= 0.
66
67
  */
67
68
  calibrationRatio?: number;
69
+ /**
70
+ * Context-fading tier persisted from a previous run's contextMeta. Seeds the
71
+ * latched cap ladder so historical tool results keep the same truncated
72
+ * bytes across runs. Invalid values start fresh; valid values clamp to the
73
+ * current context window without losing their latched provenance.
74
+ */
75
+ fadingTier?: FadingTier | null;
68
76
  /** Optional diagnostic log callback wired by the graph for observability. */
69
77
  log?: (level: 'debug' | 'info' | 'warn' | 'error', message: string, data?: Record<string, unknown>) => void;
70
78
  };
71
79
  export type PruneMessagesParams = {
80
+ /**
81
+ * Immutable graph history corresponding index-for-index with `messages`.
82
+ * When supplied, provider projections always derive from this source rather
83
+ * than from an earlier, already-truncated projection.
84
+ */
85
+ canonicalMessages?: BaseMessage[];
86
+ /**
87
+ * The caller guarantees that an existing canonical prefix cannot have been
88
+ * rewritten since the previous call. Graph reducers provide this guarantee
89
+ * by invalidating and recreating the pruner on replacements/removals.
90
+ */
91
+ canonicalPrefixStable?: boolean;
72
92
  messages: BaseMessage[];
73
93
  usageMetadata?: Partial<UsageMetadata>;
74
94
  startType?: ReturnType<BaseMessage['getType']>;
@@ -157,18 +177,47 @@ export declare function getMessagesWithinTokenLimit({ messages: _messages, maxCo
157
177
  instructionTokens?: number;
158
178
  }): PruningResult;
159
179
  export declare function checkValidNumber(value: unknown): value is number;
180
+ type FadingApplyParams = {
181
+ canonicalMessages?: BaseMessage[];
182
+ messages: BaseMessage[];
183
+ indexTokenCountMap: Record<string, number | undefined>;
184
+ tokenCounter: TokenCounter;
185
+ caps: Pick<FadingCaps, 'resultChars' | 'consumedChars' | 'inputChars'>;
186
+ /** Whether consumed results shrink to `caps.consumedChars`. */
187
+ masked: boolean;
188
+ /** First index to visit for fresh results and tool-call inputs. */
189
+ fromIndex?: number;
190
+ /** First consumed index to visit for masking. */
191
+ maskedFromIndex?: number;
192
+ /** Original (pre-masking) content keyed by message index, captured for the summarizer. */
193
+ originalContentStore?: Map<number, string>;
194
+ /** Called after storing a newly captured entry. */
195
+ onContentStored?: (index: number, content: string) => void;
196
+ };
197
+ export type FadingApplyResult = {
198
+ /** Fresh tool results rewritten. */
199
+ truncated: number;
200
+ /** Tool-call inputs rewritten. */
201
+ inputs: number;
202
+ /** Consumed tool results rewritten. */
203
+ masked: number;
204
+ /** Index of the newest AI message with text; tool results before it are consumed. */
205
+ consumedBoundary: number;
206
+ };
207
+ /**
208
+ * Applies a fading tier's caps in one forward pass. Consumed results (before
209
+ * the boundary) shrink to `consumedChars`, fresh results to `resultChars` and
210
+ * historical tool-call inputs to `inputChars`. Messages already within their
211
+ * cap keep object identity and token count, so at an unchanged tier the pass
212
+ * only touches what arrived since the watermarks. Truncation is a pure
213
+ * function of (content, cap), which is what keeps the bytes of a historical
214
+ * result identical from call to call.
215
+ */
216
+ export declare function applyFadingCaps(params: FadingApplyParams): FadingApplyResult;
160
217
  /**
161
218
  * Observation masking: replaces consumed ToolMessage content with tight
162
- * head+tail truncations that serve as informative placeholders.
163
- *
164
- * A ToolMessage is "consumed" when a subsequent AI message exists that is NOT
165
- * purely tool calls — meaning the model has already read and acted on the
166
- * result. Unconsumed results (the latest tool outputs the model hasn't
167
- * responded to yet) are left intact so the model can still use them.
168
- *
169
- * AI messages are never masked — they contain the model's own reasoning and
170
- * conclusions, which is what prevents the model from repeating work after
171
- * its tool results are masked.
219
+ * head+tail truncations that serve as informative placeholders. Fresh results
220
+ * and tool-call inputs are left alone.
172
221
  *
173
222
  * @returns The number of tool messages that were masked.
174
223
  */
@@ -176,10 +225,11 @@ export declare function maskConsumedToolResults(params: {
176
225
  messages: BaseMessage[];
177
226
  indexTokenCountMap: Record<string, number | undefined>;
178
227
  tokenCounter: TokenCounter;
179
- /** Raw-space token budget available for all consumed tool results combined.
180
- * When provided, the budget is distributed across consumed results weighted
181
- * by recency (newest get the most, oldest get MASKED_RESULT_MAX_CHARS min).
182
- * When omitted, falls back to a flat MASKED_RESULT_MAX_CHARS per result. */
228
+ /** Character cap applied to every consumed result (never below
229
+ * MASKED_RESULT_MIN_CHARS, which is also the default). */
230
+ maxChars?: number;
231
+ /** @deprecated Aggregate raw-token budget distributed by recency. Prefer
232
+ * `maxChars` for byte-stable masking across otherwise identical calls. */
183
233
  availableRawBudget?: number;
184
234
  /** When provided, original (pre-masking) content is stored here keyed by
185
235
  * message index — only for entries that actually get truncated. */
@@ -189,12 +239,8 @@ export declare function maskConsumedToolResults(params: {
189
239
  }): number;
190
240
  /**
191
241
  * Pre-flight truncation: truncates oversized ToolMessage content before the
192
- * main backward-iteration pruning runs. Unlike the ingestion guard (which caps
193
- * at tool-execution time), pre-flight truncation applies per-turn based on the
194
- * current context window budget (which may have shrunk due to growing conversation).
195
- *
196
- * After truncation, recounts tokens via tokenCounter and updates indexTokenCountMap
197
- * so subsequent pruning works with accurate counts.
242
+ * main backward-iteration pruning runs, applying one cap derived from
243
+ * `maxContextTokens` to every tool result.
198
244
  *
199
245
  * @returns The number of tool messages that were truncated.
200
246
  */
@@ -209,10 +255,8 @@ export declare function preFlightTruncateToolResults(params: {
209
255
  * invoking user-defined accessors or `toJSON`.
210
256
  */
211
257
  export declare function serializeToolCallInput(input: unknown, maxChars?: number): string;
212
- /** Per-input cap: 15% of context at ~4 chars/token, never above 200K chars. */
213
- export declare function calculateMaxToolCallInputChars(maxContextTokens?: number): number;
214
258
  /** Projects all historical tool-call input representations to bounded values. */
215
- export declare function projectToolCallInputs(messages: BaseMessage[], maxInputChars: number): BaseMessage[];
259
+ export declare function projectToolCallInputs(messages: BaseMessage[], maxInputChars: number, fromIndex?: number): BaseMessage[];
216
260
  /**
217
261
  * Derives provider-safe tool history in one pass by dropping incomplete stream
218
262
  * content and bounding every provider-consumed tool-call input representation.
@@ -234,9 +278,12 @@ export declare function createPruneMessages(factoryParams: PruneMessagesFactoryP
234
278
  originalToolContent?: Map<number, string>;
235
279
  newOriginalToolContent?: Map<number, string>;
236
280
  calibrationRatio?: number;
281
+ /** Latched fading tier after this call; hosts persist it beside calibrationRatio. */
282
+ fadingTier: FadingTier;
237
283
  resolvedInstructionOverhead?: number;
238
284
  /** Usable budget this call: maxTokens minus output reserve */
239
285
  contextBudget?: number;
240
286
  /** Calibrated instruction overhead actually applied this call */
241
287
  effectiveInstructionTokens?: number;
242
288
  };
289
+ export {};
@@ -24,6 +24,10 @@ export declare class Run<_T extends t.BaseGraphState> {
24
24
  private subagentTasks?;
25
25
  private indexTokenCountMap?;
26
26
  calibrationRatio: number;
27
+ fadingTier?: t.FadingTier;
28
+ fadingTiers: t.FadingTiers;
29
+ private fadingTierReset;
30
+ private fadingTierResetAgentIds;
27
31
  graphRunnable?: t.CompiledStateWorkflow;
28
32
  Graph: StandardGraph | MultiAgentGraph | undefined;
29
33
  returnContent: boolean;
@@ -127,6 +131,18 @@ export declare class Run<_T extends t.BaseGraphState> {
127
131
  * scaling factor instead of the default (1).
128
132
  */
129
133
  getCalibrationRatio(): number;
134
+ /**
135
+ * Returns the default agent's latched context-fading tier. Single-agent hosts
136
+ * may persist it beside the calibration ratio; multi-agent hosts should use
137
+ * `getFadingTiers()` so independent agent tiers are retained.
138
+ */
139
+ getFadingTier(): t.FadingTier | undefined;
140
+ /** Returns a defensive snapshot of latched tiers keyed by agent ID. */
141
+ getFadingTiers(): t.FadingTiers;
142
+ /** Agent IDs whose canonical history was compacted during this run. */
143
+ getFadingTierResetAgentIds(): string[];
144
+ /** Whether the default agent compacted canonical history during this run. */
145
+ didResetFadingTier(): boolean;
130
146
  getResolvedInstructionOverhead(): number | undefined;
131
147
  /**
132
148
  * Cooperative-preemption counters for this run. `emptyBoundaries` is the one
@@ -5,6 +5,13 @@ export declare class AgentSession {
5
5
  private runConfig;
6
6
  private store;
7
7
  private calibrationRatio;
8
+ private fadingTier;
9
+ private fadingTiers;
10
+ /** Reset generations keyed per scope and per tier (`''` for the default). */
11
+ private fadingGenerations;
12
+ /** Advances on every explicit history rewrite (branch, compact, restore). */
13
+ private fadingRewriteEpoch;
14
+ private alternateThreadFadingState;
8
15
  private checkpointing;
9
16
  cwd: string;
10
17
  threadId: string;
@@ -13,6 +20,14 @@ export declare class AgentSession {
13
20
  get sessionPath(): string | undefined;
14
21
  getSessionStore(): JsonlSessionStore | undefined;
15
22
  getCheckpointer(): BaseCheckpointSaver | undefined;
23
+ private getFadingState;
24
+ private setFadingState;
25
+ private trimAlternateFadingStates;
26
+ private clearFadingState;
27
+ private snapshotFadingGenerations;
28
+ private restoreFadingStateFromStore;
29
+ private persistFadingState;
30
+ private captureRunContextState;
16
31
  getLatestCheckpoint(options?: AgentSessionCheckpointLookupOptions): Promise<AgentSessionCheckpointReference | undefined>;
17
32
  private hasCheckpointState;
18
33
  private recordCheckpoint;
@@ -1,5 +1,5 @@
1
1
  import type { BaseMessage } from '@langchain/core/messages';
2
- import type { CreateSessionFileOptions, SessionBranchOptions, SessionEntry, SessionForkOptions, SessionHeader, SessionLabelEntry, SessionListItem, SessionMessageEntry, SessionCheckpointEntry, SessionCompactionEntry, SessionRunEventEntry, SessionSummaryEntry, SessionStateEntry, SessionTreeNode } from './types';
2
+ import type { CreateSessionFileOptions, SessionBranchOptions, SessionEntry, SessionForkOptions, SessionHeader, SessionLabelEntry, SessionListItem, SessionMessageEntry, SessionCheckpointEntry, SessionCompactionEntry, SessionRunEventEntry, SessionSummaryEntry, SessionStateEntry, SessionFadingState, SessionTreeNode } from './types';
3
3
  export declare class JsonlSessionStore {
4
4
  readonly path: string;
5
5
  readonly header: SessionHeader;
@@ -31,6 +31,9 @@ export declare class JsonlSessionStore {
31
31
  threadId?: string;
32
32
  }): Promise<SessionRunEventEntry>;
33
33
  setLeaf(leafId: string | null): Promise<SessionStateEntry>;
34
+ getFadingStates(mainThreadId?: string, maxAlternateStates?: number): SessionFadingState[];
35
+ hasFadingState(): boolean;
36
+ appendFadingState(fadingState: SessionStateEntry['data']['fadingState']): Promise<SessionStateEntry>;
34
37
  appendEntryForCompaction(params: {
35
38
  text: string;
36
39
  tokenCount?: number;
@@ -74,8 +74,17 @@ export type SessionRunEventEntry = SessionEntryBase<'run_event', {
74
74
  threadId?: string;
75
75
  payload?: JsonValue;
76
76
  }>;
77
+ export interface SessionFadingState {
78
+ threadId: string;
79
+ /** LangGraph checkpoint namespace within `threadId`; empty when omitted. */
80
+ checkpointNs?: string;
81
+ fadingTier?: t.FadingTier;
82
+ fadingTiers?: t.FadingTiers;
83
+ }
77
84
  export type SessionStateEntry = SessionEntryBase<'session_state', {
78
85
  leafId: string | null;
86
+ /** Latest compact fading state for one checkpoint thread; null clears all. */
87
+ fadingState?: SessionFadingState | null;
79
88
  }>;
80
89
  export type SessionEntry = SessionMessageEntry | SessionSummaryEntry | SessionCompactionEntry | SessionCheckpointEntry | SessionLabelEntry | SessionRunEventEntry | SessionStateEntry;
81
90
  export interface SessionTreeNode {
@@ -2,7 +2,7 @@ import type { ToolCall, ToolMessage } from '@langchain/core/messages/tool';
2
2
  import type { RunnableConfig } from '@langchain/core/runnables';
3
3
  import type { ToolOutputReferenceState } from '@/tools/toolOutputReferences';
4
4
  import type { ToolApprovalReplaySnapshot } from '@/hooks';
5
- import type { RunStepResumeState, ToolSessionContext } from '@/types';
5
+ import type { FadingTier, FadingTiers, RunStepResumeState, ToolSessionContext } from '@/types';
6
6
  export declare const SUBAGENT_RESUME_MANIFEST_CONFIG_KEY = "__librechat_subagent_resume_manifest";
7
7
  export declare const SUBAGENT_RESUME_ATTEMPT_CONFIG_KEY = "__librechat_subagent_resume_attempt";
8
8
  export declare const SUBAGENT_PARENT_BATCH_CONFIG_KEY = "__librechat_subagent_parent_batch";
@@ -45,6 +45,8 @@ export interface SubagentGraphResumeState {
45
45
  eagerToolSuppressions: string[];
46
46
  runStepState?: RunStepResumeState;
47
47
  toolOutputReferences?: ToolOutputReferenceState;
48
+ fadingTier?: FadingTier;
49
+ fadingTiers?: FadingTiers;
48
50
  }
49
51
  /** Private checkpoint payload linking a parent pause to an exact child state. */
50
52
  export interface SubagentResumeExecution {
@@ -74,6 +74,25 @@ export interface ContextUsageEvent {
74
74
  /** EMA ratio of provider-reported vs locally estimated token counts */
75
75
  calibrationRatio?: number;
76
76
  }
77
+ /**
78
+ * Latched context-fading tier. Caps for historical tool results derive from
79
+ * `(budgetTokens, masked)` only, so carrying the tier across runs keeps their
80
+ * truncated bytes stable for prefix-based provider prompt caches. Hosts
81
+ * persist it beside `calibrationRatio` and pass it back via
82
+ * `RunConfig.fadingTiers[agentId]`.
83
+ */
84
+ export interface FadingTier {
85
+ v: 1;
86
+ /** Token budget the caps derive from, in raw token space. Never grows;
87
+ * clamped to the current context window when seeded. */
88
+ budgetTokens: number;
89
+ /** Whether observation masking has activated. Never deactivates. */
90
+ masked: boolean;
91
+ /** Whether this tier was reduced, masked, or restored from host state. */
92
+ latched?: true;
93
+ }
94
+ /** Latched fading tiers keyed by agent ID. */
95
+ export type FadingTiers = Record<string, FadingTier>;
77
96
  export interface EventHandler {
78
97
  handle(event: string, data: StreamEventData | ModelEndData | RunStep | RunStepDeltaEvent | RunStepClosedEvent | MessageDeltaEvent | ReasoningDeltaEvent | SummarizeStartEvent | SummarizeDeltaEvent | SummarizeCompleteEvent | SubagentUpdateEvent | AgentLogEvent | ContextUsageEvent | ToolExecuteBatchRequest | {
79
98
  result: ToolEndEvent;
@@ -231,6 +250,10 @@ export type StandardGraphInput = {
231
250
  tokenCounter?: TokenCounter;
232
251
  indexTokenCountMap?: Record<string, number>;
233
252
  calibrationRatio?: number;
253
+ /** Persisted default-agent tier retained for single-agent compatibility. */
254
+ fadingTier?: FadingTier | null;
255
+ /** Persisted fading tiers keyed by agent ID. */
256
+ fadingTiers?: FadingTiers | null;
234
257
  /**
235
258
  * Receives a {@link SubagentUsageEvent} for every model call that reports
236
259
  * usage metadata inside a subagent child run spawned from this graph
@@ -319,6 +319,16 @@ export type RunConfig = {
319
319
  * conversation. Without this, the EMA resets to 1 on every new Run instance.
320
320
  */
321
321
  calibrationRatio?: number;
322
+ /**
323
+ * Default-agent fading tier retained for single-agent compatibility. New
324
+ * multi-agent integrations should use `fadingTiers`.
325
+ */
326
+ fadingTier?: g.FadingTier | null;
327
+ /**
328
+ * Context-fading tiers keyed by agent ID. Multi-agent hosts should persist
329
+ * the value returned by `Run.getFadingTiers()` and pass it back here.
330
+ */
331
+ fadingTiers?: g.FadingTiers | null;
322
332
  /** Skip post-stream cleanup (clearHeavyState) — useful for tests that inspect graph state after processStream */
323
333
  skipCleanup?: boolean;
324
334
  /**
@@ -68,3 +68,12 @@ export declare function truncateToolInput(input: unknown, maxChars: number): {
68
68
  * @returns The (possibly truncated) content string.
69
69
  */
70
70
  export declare function truncateToolResultContent(content: string, maxChars: number): string;
71
+ /** Absolute hard cap on a single tool-call input (characters). */
72
+ export declare const HARD_MAX_TOOL_CALL_INPUT_CHARS = 200000;
73
+ /** Smallest JSON value a truncated tool-call input can shrink to. */
74
+ export declare const MIN_JSON_VALUE_CHARS = 4;
75
+ /**
76
+ * Computes the max tool-call input size for a context window: 15 % of the
77
+ * window in estimated characters (~4 chars/token), capped at 200K.
78
+ */
79
+ export declare function calculateMaxToolCallInputChars(maxContextTokens?: number): number;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@librechat/agents",
3
- "version": "3.7.15",
3
+ "version": "3.7.17",
4
4
  "reova": {
5
5
  "enabled": true,
6
6
  "endpoint": "https://telemetry.reo.dev/data"
@@ -1,10 +1,15 @@
1
1
  /* eslint-disable no-console */
2
2
  import { RunnableLambda } from '@langchain/core/runnables';
3
- import { HumanMessage, SystemMessage } from '@langchain/core/messages';
3
+ import {
4
+ HumanMessage,
5
+ SystemMessage,
6
+ coerceMessageLikeToMessage,
7
+ } from '@langchain/core/messages';
4
8
  import type {
5
9
  UsageMetadata,
6
10
  BaseMessage,
7
11
  BaseMessageFields,
12
+ BaseMessageLike,
8
13
  } from '@langchain/core/messages';
9
14
  import type { RunnableConfig, Runnable } from '@langchain/core/runnables';
10
15
  import type { ExactTokenCountCache } from '@/llm/contextPressureMeter';
@@ -271,10 +276,22 @@ export class AgentContext {
271
276
  deferredToolNames: string[] = [];
272
277
  /** Running calibration ratio from the pruner — persisted across runs via contextMeta. */
273
278
  calibrationRatio: number = 1;
279
+ /** Latched context-fading tier from the pruner — persisted across runs via contextMeta. */
280
+ fadingTier?: t.FadingTier;
274
281
  /** Provider-observed instruction overhead from the pruner's best-variance turn. */
275
282
  resolvedInstructionOverhead?: number;
276
283
  private _pendingOriginalToolContent?: Map<number, string>;
277
284
  private pendingOriginalToolContentChars = 0;
285
+ /** Provider-bound projection for this Run; graph messages remain canonical. */
286
+ private providerProjectedMessages?: BaseMessage[];
287
+ /** Canonical object identities corresponding to the projected slots. */
288
+ private providerProjectionSources?: BaseMessage[];
289
+ /** Canonical messages by ID, used to distinguish identity replay from replacement. */
290
+ private providerProjectionSourcesById?: Map<string, BaseMessage>;
291
+ /** Conservative counts retained by canonical identity across a rebuild. */
292
+ private providerProjectionRecountFloors?: WeakMap<BaseMessage, number>;
293
+ /** Recount after discarding a projection whose token map contains capped sizes. */
294
+ private providerProjectionRequiresRecount = false;
278
295
  /** Pre-masking tool content keyed by message index, consumed by the summarize node. */
279
296
  get pendingOriginalToolContent(): Map<number, string> | undefined {
280
297
  return this._pendingOriginalToolContent;
@@ -390,6 +407,8 @@ export class AgentContext {
390
407
  private durableSummaryPrecedesMessages: boolean = false;
391
408
  /** Number of summarization cycles that have occurred for this agent context */
392
409
  private _summaryVersion: number = 0;
410
+ /** Whether this run compacted canonical history and reset its fading tier. */
411
+ private _fadingTierReset: boolean = false;
393
412
  /**
394
413
  * Message count at the time summarization was last triggered.
395
414
  * Used to prevent re-summarizing the same unchanged message set.
@@ -1214,6 +1233,7 @@ export class AgentContext {
1214
1233
  * Reset context for a new run
1215
1234
  */
1216
1235
  reset(options?: { preserveOriginalToolContent?: boolean }): void {
1236
+ this._fadingTierReset = false;
1217
1237
  this.systemMessageTokens = 0;
1218
1238
  this.dynamicInstructionTokens = 0;
1219
1239
  this.toolSchemaTokens = 0;
@@ -1225,6 +1245,7 @@ export class AgentContext {
1225
1245
  this.indexTokenCountMap = { ...this.baseIndexTokenCountMap };
1226
1246
  this.currentUsage = undefined;
1227
1247
  this.pruneMessages = undefined;
1248
+ this.clearProviderProjection(false);
1228
1249
  this.lastStreamCall = undefined;
1229
1250
  this.tokenTypeSwitch = undefined;
1230
1251
  this.reasoningTransitionCount = 0;
@@ -1509,9 +1530,12 @@ export class AgentContext {
1509
1530
  this._durableSummaryTokenCount = tokenCount;
1510
1531
  this.durableSummaryPrecedesMessages = this.summaryPrecedesMessages;
1511
1532
  this._summaryVersion += 1;
1533
+ this._fadingTierReset = true;
1512
1534
  this._summarizationFailures = 0;
1513
1535
  this.systemRunnableStale = true;
1514
1536
  this.pruneMessages = undefined;
1537
+ this.fadingTier = undefined;
1538
+ this.clearProviderProjection(false);
1515
1539
  }
1516
1540
 
1517
1541
  /** Sets a cross-run summary that is injected into the system prompt. */
@@ -1525,6 +1549,9 @@ export class AgentContext {
1525
1549
  this.durableSummaryPrecedesMessages = false;
1526
1550
  this._summaryVersion += 1;
1527
1551
  this.systemRunnableStale = true;
1552
+ this.pruneMessages = undefined;
1553
+ this.fadingTier = undefined;
1554
+ this.clearProviderProjection(false);
1528
1555
  }
1529
1556
 
1530
1557
  /**
@@ -1558,6 +1585,10 @@ export class AgentContext {
1558
1585
  return this._summaryVersion;
1559
1586
  }
1560
1587
 
1588
+ get fadingTierReset(): boolean {
1589
+ return this._fadingTierReset;
1590
+ }
1591
+
1561
1592
  /**
1562
1593
  * Returns true when the message count hasn't changed since the last
1563
1594
  * summarization — re-summarizing would produce an identical result.
@@ -1678,6 +1709,7 @@ export class AgentContext {
1678
1709
  this.maxContextTokens = budgetTokens;
1679
1710
  }
1680
1711
  this.pruneMessages = undefined;
1712
+ this.clearProviderProjection(true);
1681
1713
  this._lastSummarizationMsgCount = 0;
1682
1714
  this._lastOverflowPromptTokens =
1683
1715
  promptTokens != null
@@ -1686,6 +1718,147 @@ export class AgentContext {
1686
1718
  this._overflowRecoveryAttempts += 1;
1687
1719
  }
1688
1720
 
1721
+ /**
1722
+ * Returns the mutable provider projection for this Run while preserving the
1723
+ * graph-owned messages as the canonical history. Appends are synchronized
1724
+ * incrementally so the pruner's watermarks remain valid. Any rewritten or
1725
+ * compacted canonical prefix starts a fresh projection and pruner.
1726
+ */
1727
+ getProviderProjectedMessages(messages: BaseMessage[]): BaseMessage[] {
1728
+ const sources = this.providerProjectionSources;
1729
+ let projection = this.providerProjectedMessages;
1730
+ const priorLength = sources?.length ?? 0;
1731
+ const prefixChanged =
1732
+ sources != null &&
1733
+ (messages.length < sources.length ||
1734
+ (messages.length === priorLength
1735
+ ? sources.some((source, index) => messages[index] !== source)
1736
+ : priorLength > 0 &&
1737
+ messages[priorLength - 1] !== sources[priorLength - 1]));
1738
+
1739
+ if (projection == null || sources == null || prefixChanged) {
1740
+ if (prefixChanged) {
1741
+ this.pruneMessages = undefined;
1742
+ /** The originals map is keyed by index; a rewritten prefix would let
1743
+ * the summarizer restore one tool's bytes onto another message. */
1744
+ this.pendingOriginalToolContent = undefined;
1745
+ this.prepareProviderProjectionRecount(sources);
1746
+ }
1747
+ projection = messages.map((message) =>
1748
+ Array.isArray(message.content)
1749
+ ? cloneMessage(message, [...message.content])
1750
+ : message
1751
+ );
1752
+ this.providerProjectedMessages = projection;
1753
+ this.providerProjectionSources = [...messages];
1754
+ this.providerProjectionSourcesById = new Map(
1755
+ messages.flatMap((message) =>
1756
+ message.id == null ? [] : [[message.id, message] as const]
1757
+ )
1758
+ );
1759
+ if (this.providerProjectionRequiresRecount && this.tokenCounter != null) {
1760
+ const recounted: Record<string, number> = {};
1761
+ for (let i = 0; i < messages.length; i++) {
1762
+ const localCount = this.tokenCounter(messages[i]);
1763
+ const conservativeFloor =
1764
+ this.providerProjectionRecountFloors?.get(messages[i]);
1765
+ recounted[i] =
1766
+ conservativeFloor == null
1767
+ ? localCount
1768
+ : Math.max(localCount, conservativeFloor);
1769
+ }
1770
+ this.indexTokenCountMap = recounted;
1771
+ this.providerProjectionRequiresRecount = false;
1772
+ this.providerProjectionRecountFloors = undefined;
1773
+ }
1774
+ return projection;
1775
+ }
1776
+
1777
+ for (let i = sources.length; i < messages.length; i++) {
1778
+ const message = messages[i];
1779
+ sources.push(message);
1780
+ if (message.id != null) {
1781
+ this.providerProjectionSourcesById?.set(message.id, message);
1782
+ }
1783
+ projection.push(
1784
+ Array.isArray(message.content)
1785
+ ? cloneMessage(message, [...message.content])
1786
+ : message
1787
+ );
1788
+ }
1789
+ return projection;
1790
+ }
1791
+
1792
+ /**
1793
+ * Invalidates the provider projection when a graph reducer update rewrites
1794
+ * canonical history. Reducer updates expose replacements and removals before
1795
+ * they are folded into a potentially longer array, avoiding a prefix scan on
1796
+ * the normal append-only path.
1797
+ */
1798
+ invalidateProviderProjectionForMessageUpdates(
1799
+ updates:
1800
+ | BaseMessage
1801
+ | BaseMessageLike
1802
+ | Array<BaseMessage | BaseMessageLike | null | undefined>
1803
+ | null
1804
+ | undefined
1805
+ ): void {
1806
+ if (this.providerProjectionSources == null) {
1807
+ return;
1808
+ }
1809
+ const updateList = (Array.isArray(updates) ? updates : [updates]) as Array<
1810
+ BaseMessageLike | null | undefined
1811
+ >;
1812
+ for (const rawUpdate of updateList) {
1813
+ if (rawUpdate == null) {
1814
+ continue;
1815
+ }
1816
+ const update = coerceMessageLikeToMessage(rawUpdate);
1817
+ const priorSource =
1818
+ update.id == null
1819
+ ? undefined
1820
+ : this.providerProjectionSourcesById?.get(update.id);
1821
+ if (
1822
+ update.getType() === 'remove' ||
1823
+ (priorSource != null && priorSource !== update)
1824
+ ) {
1825
+ this.pruneMessages = undefined;
1826
+ this.pendingOriginalToolContent = undefined;
1827
+ this.clearProviderProjection(true);
1828
+ return;
1829
+ }
1830
+ }
1831
+ }
1832
+
1833
+ private clearProviderProjection(recountOnNextUse: boolean): void {
1834
+ if (recountOnNextUse) {
1835
+ this.prepareProviderProjectionRecount(this.providerProjectionSources);
1836
+ } else {
1837
+ this.providerProjectionRequiresRecount = false;
1838
+ this.providerProjectionRecountFloors = undefined;
1839
+ }
1840
+ this.providerProjectedMessages = undefined;
1841
+ this.providerProjectionSources = undefined;
1842
+ this.providerProjectionSourcesById = undefined;
1843
+ }
1844
+
1845
+ private prepareProviderProjectionRecount(
1846
+ sources: BaseMessage[] | undefined
1847
+ ): void {
1848
+ this.providerProjectionRequiresRecount = true;
1849
+ if (sources == null) {
1850
+ return;
1851
+ }
1852
+ const floors = new WeakMap<BaseMessage, number>();
1853
+ for (let i = 0; i < sources.length; i++) {
1854
+ const count = this.indexTokenCountMap[i];
1855
+ if (count != null && Number.isFinite(count) && count >= 0) {
1856
+ floors.set(sources[i], count);
1857
+ }
1858
+ }
1859
+ this.providerProjectionRecountFloors = floors;
1860
+ }
1861
+
1689
1862
  /** Applies token calibration only when the observation came from this provider. */
1690
1863
  applyObservedOverflowCalibration(
1691
1864
  provider: t.ProviderName | undefined,
@@ -1904,6 +2077,7 @@ export class AgentContext {
1904
2077
  summarizationEnabled: this.summarizationEnabled,
1905
2078
  reserveRatio: this.summarizationConfig?.reserveRatio,
1906
2079
  calibrationRatio: opts?.calibrationRatio ?? this.calibrationRatio,
2080
+ fadingTier: this.fadingTier,
1907
2081
  getInstructionTokens: () => this.instructionTokens,
1908
2082
  });
1909
2083
  const {
@@ -13,6 +13,8 @@ export interface ProjectAgentContextUsageParams {
13
13
  indexTokenCountMap?: Record<string, number>;
14
14
  /** Provider-calibrated ratio from a prior snapshot, applied as a static seed. */
15
15
  calibrationRatio?: number;
16
+ /** Persisted fading tier for this agent, applied to the projected branch. */
17
+ fadingTier?: t.FadingTier;
16
18
  /** Execution backend used to synthesize the live effective tool registry. */
17
19
  toolExecution?: t.ToolExecutionConfig;
18
20
  runId?: string;
@@ -34,6 +36,7 @@ export async function projectAgentContextUsage({
34
36
  tokenCounter,
35
37
  indexTokenCountMap,
36
38
  calibrationRatio,
39
+ fadingTier,
37
40
  toolExecution,
38
41
  runId,
39
42
  agentId,
@@ -44,6 +47,7 @@ export async function projectAgentContextUsage({
44
47
  indexTokenCountMap,
45
48
  toolExecution
46
49
  );
50
+ context.fadingTier = fadingTier;
47
51
  await context.tokenCalculationPromise;
48
52
  return context.projectContextUsage(messages, {
49
53
  runId,