@librechat/agents 3.7.15 → 3.7.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cjs/agents/AgentContext.cjs +91 -0
- package/dist/cjs/agents/AgentContext.cjs.map +1 -1
- package/dist/cjs/agents/projection.cjs +2 -1
- package/dist/cjs/agents/projection.cjs.map +1 -1
- package/dist/cjs/graphs/Graph.cjs +47 -6
- package/dist/cjs/graphs/Graph.cjs.map +1 -1
- package/dist/cjs/graphs/MultiAgentGraph.cjs +20 -7
- package/dist/cjs/graphs/MultiAgentGraph.cjs.map +1 -1
- package/dist/cjs/llm/anthropic/utils/message_inputs.cjs +1 -1
- package/dist/cjs/llm/openai/utils/index.cjs +1 -1
- package/dist/cjs/llm/openai/utils/index.cjs.map +1 -1
- package/dist/cjs/llm/preempt.cjs +1 -10
- package/dist/cjs/llm/preempt.cjs.map +1 -1
- package/dist/cjs/main.cjs +23 -3
- package/dist/cjs/messages/contextPruning.cjs +1 -1
- package/dist/cjs/messages/contextPruning.cjs.map +1 -1
- package/dist/cjs/messages/core.cjs +29 -1
- package/dist/cjs/messages/core.cjs.map +1 -1
- package/dist/cjs/messages/fading.cjs +122 -0
- package/dist/cjs/messages/fading.cjs.map +1 -0
- package/dist/cjs/messages/format.cjs +1 -1
- package/dist/cjs/messages/index.cjs +1 -0
- package/dist/cjs/messages/prune.cjs +364 -236
- package/dist/cjs/messages/prune.cjs.map +1 -1
- package/dist/cjs/run.cjs +27 -0
- package/dist/cjs/run.cjs.map +1 -1
- package/dist/cjs/session/AgentSession.cjs +213 -8
- package/dist/cjs/session/AgentSession.cjs.map +1 -1
- package/dist/cjs/session/JsonlSessionStore.cjs +41 -0
- package/dist/cjs/session/JsonlSessionStore.cjs.map +1 -1
- package/dist/cjs/session/index.cjs +1 -1
- package/dist/cjs/stream.cjs +1 -1
- package/dist/cjs/summarization/node.cjs +1 -1
- package/dist/cjs/tools/ToolNode.cjs +1 -1
- package/dist/cjs/tools/subagent/SubagentExecutor.cjs +2 -0
- package/dist/cjs/tools/subagent/SubagentExecutor.cjs.map +1 -1
- package/dist/cjs/tools/subagent/SubagentReplay.cjs +2 -1
- package/dist/cjs/tools/subagent/SubagentReplay.cjs.map +1 -1
- package/dist/cjs/utils/index.cjs +1 -1
- package/dist/cjs/utils/truncation.cjs +9 -0
- package/dist/cjs/utils/truncation.cjs.map +1 -1
- package/dist/esm/agents/AgentContext.mjs +92 -1
- package/dist/esm/agents/AgentContext.mjs.map +1 -1
- package/dist/esm/agents/projection.mjs +2 -1
- package/dist/esm/agents/projection.mjs.map +1 -1
- package/dist/esm/graphs/Graph.mjs +47 -6
- package/dist/esm/graphs/Graph.mjs.map +1 -1
- package/dist/esm/graphs/MultiAgentGraph.mjs +20 -7
- package/dist/esm/graphs/MultiAgentGraph.mjs.map +1 -1
- package/dist/esm/llm/anthropic/utils/message_inputs.mjs +1 -1
- package/dist/esm/llm/openai/utils/index.mjs +2 -2
- package/dist/esm/llm/openai/utils/index.mjs.map +1 -1
- package/dist/esm/llm/preempt.mjs +1 -10
- package/dist/esm/llm/preempt.mjs.map +1 -1
- package/dist/esm/main.mjs +6 -5
- package/dist/esm/messages/contextPruning.mjs +1 -1
- package/dist/esm/messages/contextPruning.mjs.map +1 -1
- package/dist/esm/messages/core.mjs +29 -2
- package/dist/esm/messages/core.mjs.map +1 -1
- package/dist/esm/messages/fading.mjs +108 -0
- package/dist/esm/messages/fading.mjs.map +1 -0
- package/dist/esm/messages/format.mjs +1 -1
- package/dist/esm/messages/index.mjs +1 -0
- package/dist/esm/messages/prune.mjs +363 -235
- package/dist/esm/messages/prune.mjs.map +1 -1
- package/dist/esm/run.mjs +27 -0
- package/dist/esm/run.mjs.map +1 -1
- package/dist/esm/session/AgentSession.mjs +213 -8
- package/dist/esm/session/AgentSession.mjs.map +1 -1
- package/dist/esm/session/JsonlSessionStore.mjs +41 -0
- package/dist/esm/session/JsonlSessionStore.mjs.map +1 -1
- package/dist/esm/session/index.mjs +1 -1
- package/dist/esm/stream.mjs +1 -1
- package/dist/esm/summarization/node.mjs +1 -1
- package/dist/esm/tools/ToolNode.mjs +1 -1
- package/dist/esm/tools/subagent/SubagentExecutor.mjs +2 -0
- package/dist/esm/tools/subagent/SubagentExecutor.mjs.map +1 -1
- package/dist/esm/tools/subagent/SubagentReplay.mjs +2 -1
- package/dist/esm/tools/subagent/SubagentReplay.mjs.map +1 -1
- package/dist/esm/utils/index.mjs +1 -1
- package/dist/esm/utils/truncation.mjs +7 -1
- package/dist/esm/utils/truncation.mjs.map +1 -1
- package/dist/types/agents/AgentContext.d.ts +32 -1
- package/dist/types/agents/projection.d.ts +3 -1
- package/dist/types/graphs/Graph.d.ts +14 -1
- package/dist/types/graphs/MultiAgentGraph.d.ts +7 -0
- package/dist/types/messages/contextPruning.d.ts +2 -0
- package/dist/types/messages/core.d.ts +2 -0
- package/dist/types/messages/fading.d.ts +81 -0
- package/dist/types/messages/index.d.ts +1 -0
- package/dist/types/messages/prune.d.ts +71 -24
- package/dist/types/run.d.ts +16 -0
- package/dist/types/session/AgentSession.d.ts +15 -0
- package/dist/types/session/JsonlSessionStore.d.ts +4 -1
- package/dist/types/session/types.d.ts +9 -0
- package/dist/types/tools/subagent/SubagentReplay.d.ts +3 -1
- package/dist/types/types/graph.d.ts +23 -0
- package/dist/types/types/run.d.ts +10 -0
- package/dist/types/utils/truncation.d.ts +9 -0
- package/package.json +1 -1
- package/src/agents/AgentContext.ts +175 -1
- package/src/agents/projection.ts +4 -0
- package/src/graphs/Graph.ts +97 -11
- package/src/graphs/MultiAgentGraph.ts +54 -6
- package/src/llm/openai/utils/index.ts +3 -1
- package/src/llm/preempt.ts +4 -17
- package/src/messages/contextPruning.ts +7 -1
- package/src/messages/core.ts +52 -0
- package/src/messages/fading.ts +301 -0
- package/src/messages/index.ts +1 -0
- package/src/messages/prune.ts +719 -507
- package/src/run.ts +52 -0
- package/src/session/AgentSession.ts +397 -8
- package/src/session/JsonlSessionStore.ts +71 -0
- package/src/session/types.ts +10 -0
- package/src/tools/subagent/SubagentExecutor.ts +6 -0
- package/src/tools/subagent/SubagentReplay.ts +15 -2
- package/src/types/graph.ts +25 -0
- package/src/types/run.ts +10 -0
- package/src/utils/truncation.ts +25 -0
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { BaseMessage, UsageMetadata } from '@langchain/core/messages';
|
|
2
|
-
import type { ContextPruningConfig } from '@/types/graph';
|
|
2
|
+
import type { ContextPruningConfig, FadingTier } from '@/types/graph';
|
|
3
|
+
import type { FadingCaps } from './fading';
|
|
3
4
|
import type { TokenCounter } from '@/types/run';
|
|
4
5
|
import type { ProviderName } from '@/types';
|
|
5
6
|
import { ContentTypes } from '@/common';
|
|
@@ -65,10 +66,29 @@ export type PruneMessagesFactoryParams = {
|
|
|
65
66
|
* of waiting for the first provider response. Ignored when <= 0.
|
|
66
67
|
*/
|
|
67
68
|
calibrationRatio?: number;
|
|
69
|
+
/**
|
|
70
|
+
* Context-fading tier persisted from a previous run's contextMeta. Seeds the
|
|
71
|
+
* latched cap ladder so historical tool results keep the same truncated
|
|
72
|
+
* bytes across runs. Invalid values start fresh; valid values clamp to the
|
|
73
|
+
* current context window without losing their latched provenance.
|
|
74
|
+
*/
|
|
75
|
+
fadingTier?: FadingTier | null;
|
|
68
76
|
/** Optional diagnostic log callback wired by the graph for observability. */
|
|
69
77
|
log?: (level: 'debug' | 'info' | 'warn' | 'error', message: string, data?: Record<string, unknown>) => void;
|
|
70
78
|
};
|
|
71
79
|
export type PruneMessagesParams = {
|
|
80
|
+
/**
|
|
81
|
+
* Immutable graph history corresponding index-for-index with `messages`.
|
|
82
|
+
* When supplied, provider projections always derive from this source rather
|
|
83
|
+
* than from an earlier, already-truncated projection.
|
|
84
|
+
*/
|
|
85
|
+
canonicalMessages?: BaseMessage[];
|
|
86
|
+
/**
|
|
87
|
+
* The caller guarantees that an existing canonical prefix cannot have been
|
|
88
|
+
* rewritten since the previous call. Graph reducers provide this guarantee
|
|
89
|
+
* by invalidating and recreating the pruner on replacements/removals.
|
|
90
|
+
*/
|
|
91
|
+
canonicalPrefixStable?: boolean;
|
|
72
92
|
messages: BaseMessage[];
|
|
73
93
|
usageMetadata?: Partial<UsageMetadata>;
|
|
74
94
|
startType?: ReturnType<BaseMessage['getType']>;
|
|
@@ -157,18 +177,47 @@ export declare function getMessagesWithinTokenLimit({ messages: _messages, maxCo
|
|
|
157
177
|
instructionTokens?: number;
|
|
158
178
|
}): PruningResult;
|
|
159
179
|
export declare function checkValidNumber(value: unknown): value is number;
|
|
180
|
+
type FadingApplyParams = {
|
|
181
|
+
canonicalMessages?: BaseMessage[];
|
|
182
|
+
messages: BaseMessage[];
|
|
183
|
+
indexTokenCountMap: Record<string, number | undefined>;
|
|
184
|
+
tokenCounter: TokenCounter;
|
|
185
|
+
caps: Pick<FadingCaps, 'resultChars' | 'consumedChars' | 'inputChars'>;
|
|
186
|
+
/** Whether consumed results shrink to `caps.consumedChars`. */
|
|
187
|
+
masked: boolean;
|
|
188
|
+
/** First index to visit for fresh results and tool-call inputs. */
|
|
189
|
+
fromIndex?: number;
|
|
190
|
+
/** First consumed index to visit for masking. */
|
|
191
|
+
maskedFromIndex?: number;
|
|
192
|
+
/** Original (pre-masking) content keyed by message index, captured for the summarizer. */
|
|
193
|
+
originalContentStore?: Map<number, string>;
|
|
194
|
+
/** Called after storing a newly captured entry. */
|
|
195
|
+
onContentStored?: (index: number, content: string) => void;
|
|
196
|
+
};
|
|
197
|
+
export type FadingApplyResult = {
|
|
198
|
+
/** Fresh tool results rewritten. */
|
|
199
|
+
truncated: number;
|
|
200
|
+
/** Tool-call inputs rewritten. */
|
|
201
|
+
inputs: number;
|
|
202
|
+
/** Consumed tool results rewritten. */
|
|
203
|
+
masked: number;
|
|
204
|
+
/** Index of the newest AI message with text; tool results before it are consumed. */
|
|
205
|
+
consumedBoundary: number;
|
|
206
|
+
};
|
|
207
|
+
/**
|
|
208
|
+
* Applies a fading tier's caps in one forward pass. Consumed results (before
|
|
209
|
+
* the boundary) shrink to `consumedChars`, fresh results to `resultChars` and
|
|
210
|
+
* historical tool-call inputs to `inputChars`. Messages already within their
|
|
211
|
+
* cap keep object identity and token count, so at an unchanged tier the pass
|
|
212
|
+
* only touches what arrived since the watermarks. Truncation is a pure
|
|
213
|
+
* function of (content, cap), which is what keeps the bytes of a historical
|
|
214
|
+
* result identical from call to call.
|
|
215
|
+
*/
|
|
216
|
+
export declare function applyFadingCaps(params: FadingApplyParams): FadingApplyResult;
|
|
160
217
|
/**
|
|
161
218
|
* Observation masking: replaces consumed ToolMessage content with tight
|
|
162
|
-
* head+tail truncations that serve as informative placeholders.
|
|
163
|
-
*
|
|
164
|
-
* A ToolMessage is "consumed" when a subsequent AI message exists that is NOT
|
|
165
|
-
* purely tool calls — meaning the model has already read and acted on the
|
|
166
|
-
* result. Unconsumed results (the latest tool outputs the model hasn't
|
|
167
|
-
* responded to yet) are left intact so the model can still use them.
|
|
168
|
-
*
|
|
169
|
-
* AI messages are never masked — they contain the model's own reasoning and
|
|
170
|
-
* conclusions, which is what prevents the model from repeating work after
|
|
171
|
-
* its tool results are masked.
|
|
219
|
+
* head+tail truncations that serve as informative placeholders. Fresh results
|
|
220
|
+
* and tool-call inputs are left alone.
|
|
172
221
|
*
|
|
173
222
|
* @returns The number of tool messages that were masked.
|
|
174
223
|
*/
|
|
@@ -176,10 +225,11 @@ export declare function maskConsumedToolResults(params: {
|
|
|
176
225
|
messages: BaseMessage[];
|
|
177
226
|
indexTokenCountMap: Record<string, number | undefined>;
|
|
178
227
|
tokenCounter: TokenCounter;
|
|
179
|
-
/**
|
|
180
|
-
*
|
|
181
|
-
|
|
182
|
-
|
|
228
|
+
/** Character cap applied to every consumed result (never below
|
|
229
|
+
* MASKED_RESULT_MIN_CHARS, which is also the default). */
|
|
230
|
+
maxChars?: number;
|
|
231
|
+
/** @deprecated Aggregate raw-token budget distributed by recency. Prefer
|
|
232
|
+
* `maxChars` for byte-stable masking across otherwise identical calls. */
|
|
183
233
|
availableRawBudget?: number;
|
|
184
234
|
/** When provided, original (pre-masking) content is stored here keyed by
|
|
185
235
|
* message index — only for entries that actually get truncated. */
|
|
@@ -189,12 +239,8 @@ export declare function maskConsumedToolResults(params: {
|
|
|
189
239
|
}): number;
|
|
190
240
|
/**
|
|
191
241
|
* Pre-flight truncation: truncates oversized ToolMessage content before the
|
|
192
|
-
* main backward-iteration pruning runs
|
|
193
|
-
*
|
|
194
|
-
* current context window budget (which may have shrunk due to growing conversation).
|
|
195
|
-
*
|
|
196
|
-
* After truncation, recounts tokens via tokenCounter and updates indexTokenCountMap
|
|
197
|
-
* so subsequent pruning works with accurate counts.
|
|
242
|
+
* main backward-iteration pruning runs, applying one cap derived from
|
|
243
|
+
* `maxContextTokens` to every tool result.
|
|
198
244
|
*
|
|
199
245
|
* @returns The number of tool messages that were truncated.
|
|
200
246
|
*/
|
|
@@ -209,10 +255,8 @@ export declare function preFlightTruncateToolResults(params: {
|
|
|
209
255
|
* invoking user-defined accessors or `toJSON`.
|
|
210
256
|
*/
|
|
211
257
|
export declare function serializeToolCallInput(input: unknown, maxChars?: number): string;
|
|
212
|
-
/** Per-input cap: 15% of context at ~4 chars/token, never above 200K chars. */
|
|
213
|
-
export declare function calculateMaxToolCallInputChars(maxContextTokens?: number): number;
|
|
214
258
|
/** Projects all historical tool-call input representations to bounded values. */
|
|
215
|
-
export declare function projectToolCallInputs(messages: BaseMessage[], maxInputChars: number): BaseMessage[];
|
|
259
|
+
export declare function projectToolCallInputs(messages: BaseMessage[], maxInputChars: number, fromIndex?: number): BaseMessage[];
|
|
216
260
|
/**
|
|
217
261
|
* Derives provider-safe tool history in one pass by dropping incomplete stream
|
|
218
262
|
* content and bounding every provider-consumed tool-call input representation.
|
|
@@ -234,9 +278,12 @@ export declare function createPruneMessages(factoryParams: PruneMessagesFactoryP
|
|
|
234
278
|
originalToolContent?: Map<number, string>;
|
|
235
279
|
newOriginalToolContent?: Map<number, string>;
|
|
236
280
|
calibrationRatio?: number;
|
|
281
|
+
/** Latched fading tier after this call; hosts persist it beside calibrationRatio. */
|
|
282
|
+
fadingTier: FadingTier;
|
|
237
283
|
resolvedInstructionOverhead?: number;
|
|
238
284
|
/** Usable budget this call: maxTokens minus output reserve */
|
|
239
285
|
contextBudget?: number;
|
|
240
286
|
/** Calibrated instruction overhead actually applied this call */
|
|
241
287
|
effectiveInstructionTokens?: number;
|
|
242
288
|
};
|
|
289
|
+
export {};
|
package/dist/types/run.d.ts
CHANGED
|
@@ -24,6 +24,10 @@ export declare class Run<_T extends t.BaseGraphState> {
|
|
|
24
24
|
private subagentTasks?;
|
|
25
25
|
private indexTokenCountMap?;
|
|
26
26
|
calibrationRatio: number;
|
|
27
|
+
fadingTier?: t.FadingTier;
|
|
28
|
+
fadingTiers: t.FadingTiers;
|
|
29
|
+
private fadingTierReset;
|
|
30
|
+
private fadingTierResetAgentIds;
|
|
27
31
|
graphRunnable?: t.CompiledStateWorkflow;
|
|
28
32
|
Graph: StandardGraph | MultiAgentGraph | undefined;
|
|
29
33
|
returnContent: boolean;
|
|
@@ -127,6 +131,18 @@ export declare class Run<_T extends t.BaseGraphState> {
|
|
|
127
131
|
* scaling factor instead of the default (1).
|
|
128
132
|
*/
|
|
129
133
|
getCalibrationRatio(): number;
|
|
134
|
+
/**
|
|
135
|
+
* Returns the default agent's latched context-fading tier. Single-agent hosts
|
|
136
|
+
* may persist it beside the calibration ratio; multi-agent hosts should use
|
|
137
|
+
* `getFadingTiers()` so independent agent tiers are retained.
|
|
138
|
+
*/
|
|
139
|
+
getFadingTier(): t.FadingTier | undefined;
|
|
140
|
+
/** Returns a defensive snapshot of latched tiers keyed by agent ID. */
|
|
141
|
+
getFadingTiers(): t.FadingTiers;
|
|
142
|
+
/** Agent IDs whose canonical history was compacted during this run. */
|
|
143
|
+
getFadingTierResetAgentIds(): string[];
|
|
144
|
+
/** Whether the default agent compacted canonical history during this run. */
|
|
145
|
+
didResetFadingTier(): boolean;
|
|
130
146
|
getResolvedInstructionOverhead(): number | undefined;
|
|
131
147
|
/**
|
|
132
148
|
* Cooperative-preemption counters for this run. `emptyBoundaries` is the one
|
|
@@ -5,6 +5,13 @@ export declare class AgentSession {
|
|
|
5
5
|
private runConfig;
|
|
6
6
|
private store;
|
|
7
7
|
private calibrationRatio;
|
|
8
|
+
private fadingTier;
|
|
9
|
+
private fadingTiers;
|
|
10
|
+
/** Reset generations keyed per scope and per tier (`''` for the default). */
|
|
11
|
+
private fadingGenerations;
|
|
12
|
+
/** Advances on every explicit history rewrite (branch, compact, restore). */
|
|
13
|
+
private fadingRewriteEpoch;
|
|
14
|
+
private alternateThreadFadingState;
|
|
8
15
|
private checkpointing;
|
|
9
16
|
cwd: string;
|
|
10
17
|
threadId: string;
|
|
@@ -13,6 +20,14 @@ export declare class AgentSession {
|
|
|
13
20
|
get sessionPath(): string | undefined;
|
|
14
21
|
getSessionStore(): JsonlSessionStore | undefined;
|
|
15
22
|
getCheckpointer(): BaseCheckpointSaver | undefined;
|
|
23
|
+
private getFadingState;
|
|
24
|
+
private setFadingState;
|
|
25
|
+
private trimAlternateFadingStates;
|
|
26
|
+
private clearFadingState;
|
|
27
|
+
private snapshotFadingGenerations;
|
|
28
|
+
private restoreFadingStateFromStore;
|
|
29
|
+
private persistFadingState;
|
|
30
|
+
private captureRunContextState;
|
|
16
31
|
getLatestCheckpoint(options?: AgentSessionCheckpointLookupOptions): Promise<AgentSessionCheckpointReference | undefined>;
|
|
17
32
|
private hasCheckpointState;
|
|
18
33
|
private recordCheckpoint;
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { BaseMessage } from '@langchain/core/messages';
|
|
2
|
-
import type { CreateSessionFileOptions, SessionBranchOptions, SessionEntry, SessionForkOptions, SessionHeader, SessionLabelEntry, SessionListItem, SessionMessageEntry, SessionCheckpointEntry, SessionCompactionEntry, SessionRunEventEntry, SessionSummaryEntry, SessionStateEntry, SessionTreeNode } from './types';
|
|
2
|
+
import type { CreateSessionFileOptions, SessionBranchOptions, SessionEntry, SessionForkOptions, SessionHeader, SessionLabelEntry, SessionListItem, SessionMessageEntry, SessionCheckpointEntry, SessionCompactionEntry, SessionRunEventEntry, SessionSummaryEntry, SessionStateEntry, SessionFadingState, SessionTreeNode } from './types';
|
|
3
3
|
export declare class JsonlSessionStore {
|
|
4
4
|
readonly path: string;
|
|
5
5
|
readonly header: SessionHeader;
|
|
@@ -31,6 +31,9 @@ export declare class JsonlSessionStore {
|
|
|
31
31
|
threadId?: string;
|
|
32
32
|
}): Promise<SessionRunEventEntry>;
|
|
33
33
|
setLeaf(leafId: string | null): Promise<SessionStateEntry>;
|
|
34
|
+
getFadingStates(mainThreadId?: string, maxAlternateStates?: number): SessionFadingState[];
|
|
35
|
+
hasFadingState(): boolean;
|
|
36
|
+
appendFadingState(fadingState: SessionStateEntry['data']['fadingState']): Promise<SessionStateEntry>;
|
|
34
37
|
appendEntryForCompaction(params: {
|
|
35
38
|
text: string;
|
|
36
39
|
tokenCount?: number;
|
|
@@ -74,8 +74,17 @@ export type SessionRunEventEntry = SessionEntryBase<'run_event', {
|
|
|
74
74
|
threadId?: string;
|
|
75
75
|
payload?: JsonValue;
|
|
76
76
|
}>;
|
|
77
|
+
export interface SessionFadingState {
|
|
78
|
+
threadId: string;
|
|
79
|
+
/** LangGraph checkpoint namespace within `threadId`; empty when omitted. */
|
|
80
|
+
checkpointNs?: string;
|
|
81
|
+
fadingTier?: t.FadingTier;
|
|
82
|
+
fadingTiers?: t.FadingTiers;
|
|
83
|
+
}
|
|
77
84
|
export type SessionStateEntry = SessionEntryBase<'session_state', {
|
|
78
85
|
leafId: string | null;
|
|
86
|
+
/** Latest compact fading state for one checkpoint thread; null clears all. */
|
|
87
|
+
fadingState?: SessionFadingState | null;
|
|
79
88
|
}>;
|
|
80
89
|
export type SessionEntry = SessionMessageEntry | SessionSummaryEntry | SessionCompactionEntry | SessionCheckpointEntry | SessionLabelEntry | SessionRunEventEntry | SessionStateEntry;
|
|
81
90
|
export interface SessionTreeNode {
|
|
@@ -2,7 +2,7 @@ import type { ToolCall, ToolMessage } from '@langchain/core/messages/tool';
|
|
|
2
2
|
import type { RunnableConfig } from '@langchain/core/runnables';
|
|
3
3
|
import type { ToolOutputReferenceState } from '@/tools/toolOutputReferences';
|
|
4
4
|
import type { ToolApprovalReplaySnapshot } from '@/hooks';
|
|
5
|
-
import type { RunStepResumeState, ToolSessionContext } from '@/types';
|
|
5
|
+
import type { FadingTier, FadingTiers, RunStepResumeState, ToolSessionContext } from '@/types';
|
|
6
6
|
export declare const SUBAGENT_RESUME_MANIFEST_CONFIG_KEY = "__librechat_subagent_resume_manifest";
|
|
7
7
|
export declare const SUBAGENT_RESUME_ATTEMPT_CONFIG_KEY = "__librechat_subagent_resume_attempt";
|
|
8
8
|
export declare const SUBAGENT_PARENT_BATCH_CONFIG_KEY = "__librechat_subagent_parent_batch";
|
|
@@ -45,6 +45,8 @@ export interface SubagentGraphResumeState {
|
|
|
45
45
|
eagerToolSuppressions: string[];
|
|
46
46
|
runStepState?: RunStepResumeState;
|
|
47
47
|
toolOutputReferences?: ToolOutputReferenceState;
|
|
48
|
+
fadingTier?: FadingTier;
|
|
49
|
+
fadingTiers?: FadingTiers;
|
|
48
50
|
}
|
|
49
51
|
/** Private checkpoint payload linking a parent pause to an exact child state. */
|
|
50
52
|
export interface SubagentResumeExecution {
|
|
@@ -74,6 +74,25 @@ export interface ContextUsageEvent {
|
|
|
74
74
|
/** EMA ratio of provider-reported vs locally estimated token counts */
|
|
75
75
|
calibrationRatio?: number;
|
|
76
76
|
}
|
|
77
|
+
/**
|
|
78
|
+
* Latched context-fading tier. Caps for historical tool results derive from
|
|
79
|
+
* `(budgetTokens, masked)` only, so carrying the tier across runs keeps their
|
|
80
|
+
* truncated bytes stable for prefix-based provider prompt caches. Hosts
|
|
81
|
+
* persist it beside `calibrationRatio` and pass it back via
|
|
82
|
+
* `RunConfig.fadingTiers[agentId]`.
|
|
83
|
+
*/
|
|
84
|
+
export interface FadingTier {
|
|
85
|
+
v: 1;
|
|
86
|
+
/** Token budget the caps derive from, in raw token space. Never grows;
|
|
87
|
+
* clamped to the current context window when seeded. */
|
|
88
|
+
budgetTokens: number;
|
|
89
|
+
/** Whether observation masking has activated. Never deactivates. */
|
|
90
|
+
masked: boolean;
|
|
91
|
+
/** Whether this tier was reduced, masked, or restored from host state. */
|
|
92
|
+
latched?: true;
|
|
93
|
+
}
|
|
94
|
+
/** Latched fading tiers keyed by agent ID. */
|
|
95
|
+
export type FadingTiers = Record<string, FadingTier>;
|
|
77
96
|
export interface EventHandler {
|
|
78
97
|
handle(event: string, data: StreamEventData | ModelEndData | RunStep | RunStepDeltaEvent | RunStepClosedEvent | MessageDeltaEvent | ReasoningDeltaEvent | SummarizeStartEvent | SummarizeDeltaEvent | SummarizeCompleteEvent | SubagentUpdateEvent | AgentLogEvent | ContextUsageEvent | ToolExecuteBatchRequest | {
|
|
79
98
|
result: ToolEndEvent;
|
|
@@ -231,6 +250,10 @@ export type StandardGraphInput = {
|
|
|
231
250
|
tokenCounter?: TokenCounter;
|
|
232
251
|
indexTokenCountMap?: Record<string, number>;
|
|
233
252
|
calibrationRatio?: number;
|
|
253
|
+
/** Persisted default-agent tier retained for single-agent compatibility. */
|
|
254
|
+
fadingTier?: FadingTier | null;
|
|
255
|
+
/** Persisted fading tiers keyed by agent ID. */
|
|
256
|
+
fadingTiers?: FadingTiers | null;
|
|
234
257
|
/**
|
|
235
258
|
* Receives a {@link SubagentUsageEvent} for every model call that reports
|
|
236
259
|
* usage metadata inside a subagent child run spawned from this graph
|
|
@@ -319,6 +319,16 @@ export type RunConfig = {
|
|
|
319
319
|
* conversation. Without this, the EMA resets to 1 on every new Run instance.
|
|
320
320
|
*/
|
|
321
321
|
calibrationRatio?: number;
|
|
322
|
+
/**
|
|
323
|
+
* Default-agent fading tier retained for single-agent compatibility. New
|
|
324
|
+
* multi-agent integrations should use `fadingTiers`.
|
|
325
|
+
*/
|
|
326
|
+
fadingTier?: g.FadingTier | null;
|
|
327
|
+
/**
|
|
328
|
+
* Context-fading tiers keyed by agent ID. Multi-agent hosts should persist
|
|
329
|
+
* the value returned by `Run.getFadingTiers()` and pass it back here.
|
|
330
|
+
*/
|
|
331
|
+
fadingTiers?: g.FadingTiers | null;
|
|
322
332
|
/** Skip post-stream cleanup (clearHeavyState) — useful for tests that inspect graph state after processStream */
|
|
323
333
|
skipCleanup?: boolean;
|
|
324
334
|
/**
|
|
@@ -68,3 +68,12 @@ export declare function truncateToolInput(input: unknown, maxChars: number): {
|
|
|
68
68
|
* @returns The (possibly truncated) content string.
|
|
69
69
|
*/
|
|
70
70
|
export declare function truncateToolResultContent(content: string, maxChars: number): string;
|
|
71
|
+
/** Absolute hard cap on a single tool-call input (characters). */
|
|
72
|
+
export declare const HARD_MAX_TOOL_CALL_INPUT_CHARS = 200000;
|
|
73
|
+
/** Smallest JSON value a truncated tool-call input can shrink to. */
|
|
74
|
+
export declare const MIN_JSON_VALUE_CHARS = 4;
|
|
75
|
+
/**
|
|
76
|
+
* Computes the max tool-call input size for a context window: 15 % of the
|
|
77
|
+
* window in estimated characters (~4 chars/token), capped at 200K.
|
|
78
|
+
*/
|
|
79
|
+
export declare function calculateMaxToolCallInputChars(maxContextTokens?: number): number;
|
package/package.json
CHANGED
|
@@ -1,10 +1,15 @@
|
|
|
1
1
|
/* eslint-disable no-console */
|
|
2
2
|
import { RunnableLambda } from '@langchain/core/runnables';
|
|
3
|
-
import {
|
|
3
|
+
import {
|
|
4
|
+
HumanMessage,
|
|
5
|
+
SystemMessage,
|
|
6
|
+
coerceMessageLikeToMessage,
|
|
7
|
+
} from '@langchain/core/messages';
|
|
4
8
|
import type {
|
|
5
9
|
UsageMetadata,
|
|
6
10
|
BaseMessage,
|
|
7
11
|
BaseMessageFields,
|
|
12
|
+
BaseMessageLike,
|
|
8
13
|
} from '@langchain/core/messages';
|
|
9
14
|
import type { RunnableConfig, Runnable } from '@langchain/core/runnables';
|
|
10
15
|
import type { ExactTokenCountCache } from '@/llm/contextPressureMeter';
|
|
@@ -271,10 +276,22 @@ export class AgentContext {
|
|
|
271
276
|
deferredToolNames: string[] = [];
|
|
272
277
|
/** Running calibration ratio from the pruner — persisted across runs via contextMeta. */
|
|
273
278
|
calibrationRatio: number = 1;
|
|
279
|
+
/** Latched context-fading tier from the pruner — persisted across runs via contextMeta. */
|
|
280
|
+
fadingTier?: t.FadingTier;
|
|
274
281
|
/** Provider-observed instruction overhead from the pruner's best-variance turn. */
|
|
275
282
|
resolvedInstructionOverhead?: number;
|
|
276
283
|
private _pendingOriginalToolContent?: Map<number, string>;
|
|
277
284
|
private pendingOriginalToolContentChars = 0;
|
|
285
|
+
/** Provider-bound projection for this Run; graph messages remain canonical. */
|
|
286
|
+
private providerProjectedMessages?: BaseMessage[];
|
|
287
|
+
/** Canonical object identities corresponding to the projected slots. */
|
|
288
|
+
private providerProjectionSources?: BaseMessage[];
|
|
289
|
+
/** Canonical messages by ID, used to distinguish identity replay from replacement. */
|
|
290
|
+
private providerProjectionSourcesById?: Map<string, BaseMessage>;
|
|
291
|
+
/** Conservative counts retained by canonical identity across a rebuild. */
|
|
292
|
+
private providerProjectionRecountFloors?: WeakMap<BaseMessage, number>;
|
|
293
|
+
/** Recount after discarding a projection whose token map contains capped sizes. */
|
|
294
|
+
private providerProjectionRequiresRecount = false;
|
|
278
295
|
/** Pre-masking tool content keyed by message index, consumed by the summarize node. */
|
|
279
296
|
get pendingOriginalToolContent(): Map<number, string> | undefined {
|
|
280
297
|
return this._pendingOriginalToolContent;
|
|
@@ -390,6 +407,8 @@ export class AgentContext {
|
|
|
390
407
|
private durableSummaryPrecedesMessages: boolean = false;
|
|
391
408
|
/** Number of summarization cycles that have occurred for this agent context */
|
|
392
409
|
private _summaryVersion: number = 0;
|
|
410
|
+
/** Whether this run compacted canonical history and reset its fading tier. */
|
|
411
|
+
private _fadingTierReset: boolean = false;
|
|
393
412
|
/**
|
|
394
413
|
* Message count at the time summarization was last triggered.
|
|
395
414
|
* Used to prevent re-summarizing the same unchanged message set.
|
|
@@ -1214,6 +1233,7 @@ export class AgentContext {
|
|
|
1214
1233
|
* Reset context for a new run
|
|
1215
1234
|
*/
|
|
1216
1235
|
reset(options?: { preserveOriginalToolContent?: boolean }): void {
|
|
1236
|
+
this._fadingTierReset = false;
|
|
1217
1237
|
this.systemMessageTokens = 0;
|
|
1218
1238
|
this.dynamicInstructionTokens = 0;
|
|
1219
1239
|
this.toolSchemaTokens = 0;
|
|
@@ -1225,6 +1245,7 @@ export class AgentContext {
|
|
|
1225
1245
|
this.indexTokenCountMap = { ...this.baseIndexTokenCountMap };
|
|
1226
1246
|
this.currentUsage = undefined;
|
|
1227
1247
|
this.pruneMessages = undefined;
|
|
1248
|
+
this.clearProviderProjection(false);
|
|
1228
1249
|
this.lastStreamCall = undefined;
|
|
1229
1250
|
this.tokenTypeSwitch = undefined;
|
|
1230
1251
|
this.reasoningTransitionCount = 0;
|
|
@@ -1509,9 +1530,12 @@ export class AgentContext {
|
|
|
1509
1530
|
this._durableSummaryTokenCount = tokenCount;
|
|
1510
1531
|
this.durableSummaryPrecedesMessages = this.summaryPrecedesMessages;
|
|
1511
1532
|
this._summaryVersion += 1;
|
|
1533
|
+
this._fadingTierReset = true;
|
|
1512
1534
|
this._summarizationFailures = 0;
|
|
1513
1535
|
this.systemRunnableStale = true;
|
|
1514
1536
|
this.pruneMessages = undefined;
|
|
1537
|
+
this.fadingTier = undefined;
|
|
1538
|
+
this.clearProviderProjection(false);
|
|
1515
1539
|
}
|
|
1516
1540
|
|
|
1517
1541
|
/** Sets a cross-run summary that is injected into the system prompt. */
|
|
@@ -1525,6 +1549,9 @@ export class AgentContext {
|
|
|
1525
1549
|
this.durableSummaryPrecedesMessages = false;
|
|
1526
1550
|
this._summaryVersion += 1;
|
|
1527
1551
|
this.systemRunnableStale = true;
|
|
1552
|
+
this.pruneMessages = undefined;
|
|
1553
|
+
this.fadingTier = undefined;
|
|
1554
|
+
this.clearProviderProjection(false);
|
|
1528
1555
|
}
|
|
1529
1556
|
|
|
1530
1557
|
/**
|
|
@@ -1558,6 +1585,10 @@ export class AgentContext {
|
|
|
1558
1585
|
return this._summaryVersion;
|
|
1559
1586
|
}
|
|
1560
1587
|
|
|
1588
|
+
get fadingTierReset(): boolean {
|
|
1589
|
+
return this._fadingTierReset;
|
|
1590
|
+
}
|
|
1591
|
+
|
|
1561
1592
|
/**
|
|
1562
1593
|
* Returns true when the message count hasn't changed since the last
|
|
1563
1594
|
* summarization — re-summarizing would produce an identical result.
|
|
@@ -1678,6 +1709,7 @@ export class AgentContext {
|
|
|
1678
1709
|
this.maxContextTokens = budgetTokens;
|
|
1679
1710
|
}
|
|
1680
1711
|
this.pruneMessages = undefined;
|
|
1712
|
+
this.clearProviderProjection(true);
|
|
1681
1713
|
this._lastSummarizationMsgCount = 0;
|
|
1682
1714
|
this._lastOverflowPromptTokens =
|
|
1683
1715
|
promptTokens != null
|
|
@@ -1686,6 +1718,147 @@ export class AgentContext {
|
|
|
1686
1718
|
this._overflowRecoveryAttempts += 1;
|
|
1687
1719
|
}
|
|
1688
1720
|
|
|
1721
|
+
/**
|
|
1722
|
+
* Returns the mutable provider projection for this Run while preserving the
|
|
1723
|
+
* graph-owned messages as the canonical history. Appends are synchronized
|
|
1724
|
+
* incrementally so the pruner's watermarks remain valid. Any rewritten or
|
|
1725
|
+
* compacted canonical prefix starts a fresh projection and pruner.
|
|
1726
|
+
*/
|
|
1727
|
+
getProviderProjectedMessages(messages: BaseMessage[]): BaseMessage[] {
|
|
1728
|
+
const sources = this.providerProjectionSources;
|
|
1729
|
+
let projection = this.providerProjectedMessages;
|
|
1730
|
+
const priorLength = sources?.length ?? 0;
|
|
1731
|
+
const prefixChanged =
|
|
1732
|
+
sources != null &&
|
|
1733
|
+
(messages.length < sources.length ||
|
|
1734
|
+
(messages.length === priorLength
|
|
1735
|
+
? sources.some((source, index) => messages[index] !== source)
|
|
1736
|
+
: priorLength > 0 &&
|
|
1737
|
+
messages[priorLength - 1] !== sources[priorLength - 1]));
|
|
1738
|
+
|
|
1739
|
+
if (projection == null || sources == null || prefixChanged) {
|
|
1740
|
+
if (prefixChanged) {
|
|
1741
|
+
this.pruneMessages = undefined;
|
|
1742
|
+
/** The originals map is keyed by index; a rewritten prefix would let
|
|
1743
|
+
* the summarizer restore one tool's bytes onto another message. */
|
|
1744
|
+
this.pendingOriginalToolContent = undefined;
|
|
1745
|
+
this.prepareProviderProjectionRecount(sources);
|
|
1746
|
+
}
|
|
1747
|
+
projection = messages.map((message) =>
|
|
1748
|
+
Array.isArray(message.content)
|
|
1749
|
+
? cloneMessage(message, [...message.content])
|
|
1750
|
+
: message
|
|
1751
|
+
);
|
|
1752
|
+
this.providerProjectedMessages = projection;
|
|
1753
|
+
this.providerProjectionSources = [...messages];
|
|
1754
|
+
this.providerProjectionSourcesById = new Map(
|
|
1755
|
+
messages.flatMap((message) =>
|
|
1756
|
+
message.id == null ? [] : [[message.id, message] as const]
|
|
1757
|
+
)
|
|
1758
|
+
);
|
|
1759
|
+
if (this.providerProjectionRequiresRecount && this.tokenCounter != null) {
|
|
1760
|
+
const recounted: Record<string, number> = {};
|
|
1761
|
+
for (let i = 0; i < messages.length; i++) {
|
|
1762
|
+
const localCount = this.tokenCounter(messages[i]);
|
|
1763
|
+
const conservativeFloor =
|
|
1764
|
+
this.providerProjectionRecountFloors?.get(messages[i]);
|
|
1765
|
+
recounted[i] =
|
|
1766
|
+
conservativeFloor == null
|
|
1767
|
+
? localCount
|
|
1768
|
+
: Math.max(localCount, conservativeFloor);
|
|
1769
|
+
}
|
|
1770
|
+
this.indexTokenCountMap = recounted;
|
|
1771
|
+
this.providerProjectionRequiresRecount = false;
|
|
1772
|
+
this.providerProjectionRecountFloors = undefined;
|
|
1773
|
+
}
|
|
1774
|
+
return projection;
|
|
1775
|
+
}
|
|
1776
|
+
|
|
1777
|
+
for (let i = sources.length; i < messages.length; i++) {
|
|
1778
|
+
const message = messages[i];
|
|
1779
|
+
sources.push(message);
|
|
1780
|
+
if (message.id != null) {
|
|
1781
|
+
this.providerProjectionSourcesById?.set(message.id, message);
|
|
1782
|
+
}
|
|
1783
|
+
projection.push(
|
|
1784
|
+
Array.isArray(message.content)
|
|
1785
|
+
? cloneMessage(message, [...message.content])
|
|
1786
|
+
: message
|
|
1787
|
+
);
|
|
1788
|
+
}
|
|
1789
|
+
return projection;
|
|
1790
|
+
}
|
|
1791
|
+
|
|
1792
|
+
/**
|
|
1793
|
+
* Invalidates the provider projection when a graph reducer update rewrites
|
|
1794
|
+
* canonical history. Reducer updates expose replacements and removals before
|
|
1795
|
+
* they are folded into a potentially longer array, avoiding a prefix scan on
|
|
1796
|
+
* the normal append-only path.
|
|
1797
|
+
*/
|
|
1798
|
+
invalidateProviderProjectionForMessageUpdates(
|
|
1799
|
+
updates:
|
|
1800
|
+
| BaseMessage
|
|
1801
|
+
| BaseMessageLike
|
|
1802
|
+
| Array<BaseMessage | BaseMessageLike | null | undefined>
|
|
1803
|
+
| null
|
|
1804
|
+
| undefined
|
|
1805
|
+
): void {
|
|
1806
|
+
if (this.providerProjectionSources == null) {
|
|
1807
|
+
return;
|
|
1808
|
+
}
|
|
1809
|
+
const updateList = (Array.isArray(updates) ? updates : [updates]) as Array<
|
|
1810
|
+
BaseMessageLike | null | undefined
|
|
1811
|
+
>;
|
|
1812
|
+
for (const rawUpdate of updateList) {
|
|
1813
|
+
if (rawUpdate == null) {
|
|
1814
|
+
continue;
|
|
1815
|
+
}
|
|
1816
|
+
const update = coerceMessageLikeToMessage(rawUpdate);
|
|
1817
|
+
const priorSource =
|
|
1818
|
+
update.id == null
|
|
1819
|
+
? undefined
|
|
1820
|
+
: this.providerProjectionSourcesById?.get(update.id);
|
|
1821
|
+
if (
|
|
1822
|
+
update.getType() === 'remove' ||
|
|
1823
|
+
(priorSource != null && priorSource !== update)
|
|
1824
|
+
) {
|
|
1825
|
+
this.pruneMessages = undefined;
|
|
1826
|
+
this.pendingOriginalToolContent = undefined;
|
|
1827
|
+
this.clearProviderProjection(true);
|
|
1828
|
+
return;
|
|
1829
|
+
}
|
|
1830
|
+
}
|
|
1831
|
+
}
|
|
1832
|
+
|
|
1833
|
+
private clearProviderProjection(recountOnNextUse: boolean): void {
|
|
1834
|
+
if (recountOnNextUse) {
|
|
1835
|
+
this.prepareProviderProjectionRecount(this.providerProjectionSources);
|
|
1836
|
+
} else {
|
|
1837
|
+
this.providerProjectionRequiresRecount = false;
|
|
1838
|
+
this.providerProjectionRecountFloors = undefined;
|
|
1839
|
+
}
|
|
1840
|
+
this.providerProjectedMessages = undefined;
|
|
1841
|
+
this.providerProjectionSources = undefined;
|
|
1842
|
+
this.providerProjectionSourcesById = undefined;
|
|
1843
|
+
}
|
|
1844
|
+
|
|
1845
|
+
private prepareProviderProjectionRecount(
|
|
1846
|
+
sources: BaseMessage[] | undefined
|
|
1847
|
+
): void {
|
|
1848
|
+
this.providerProjectionRequiresRecount = true;
|
|
1849
|
+
if (sources == null) {
|
|
1850
|
+
return;
|
|
1851
|
+
}
|
|
1852
|
+
const floors = new WeakMap<BaseMessage, number>();
|
|
1853
|
+
for (let i = 0; i < sources.length; i++) {
|
|
1854
|
+
const count = this.indexTokenCountMap[i];
|
|
1855
|
+
if (count != null && Number.isFinite(count) && count >= 0) {
|
|
1856
|
+
floors.set(sources[i], count);
|
|
1857
|
+
}
|
|
1858
|
+
}
|
|
1859
|
+
this.providerProjectionRecountFloors = floors;
|
|
1860
|
+
}
|
|
1861
|
+
|
|
1689
1862
|
/** Applies token calibration only when the observation came from this provider. */
|
|
1690
1863
|
applyObservedOverflowCalibration(
|
|
1691
1864
|
provider: t.ProviderName | undefined,
|
|
@@ -1904,6 +2077,7 @@ export class AgentContext {
|
|
|
1904
2077
|
summarizationEnabled: this.summarizationEnabled,
|
|
1905
2078
|
reserveRatio: this.summarizationConfig?.reserveRatio,
|
|
1906
2079
|
calibrationRatio: opts?.calibrationRatio ?? this.calibrationRatio,
|
|
2080
|
+
fadingTier: this.fadingTier,
|
|
1907
2081
|
getInstructionTokens: () => this.instructionTokens,
|
|
1908
2082
|
});
|
|
1909
2083
|
const {
|
package/src/agents/projection.ts
CHANGED
|
@@ -13,6 +13,8 @@ export interface ProjectAgentContextUsageParams {
|
|
|
13
13
|
indexTokenCountMap?: Record<string, number>;
|
|
14
14
|
/** Provider-calibrated ratio from a prior snapshot, applied as a static seed. */
|
|
15
15
|
calibrationRatio?: number;
|
|
16
|
+
/** Persisted fading tier for this agent, applied to the projected branch. */
|
|
17
|
+
fadingTier?: t.FadingTier;
|
|
16
18
|
/** Execution backend used to synthesize the live effective tool registry. */
|
|
17
19
|
toolExecution?: t.ToolExecutionConfig;
|
|
18
20
|
runId?: string;
|
|
@@ -34,6 +36,7 @@ export async function projectAgentContextUsage({
|
|
|
34
36
|
tokenCounter,
|
|
35
37
|
indexTokenCountMap,
|
|
36
38
|
calibrationRatio,
|
|
39
|
+
fadingTier,
|
|
37
40
|
toolExecution,
|
|
38
41
|
runId,
|
|
39
42
|
agentId,
|
|
@@ -44,6 +47,7 @@ export async function projectAgentContextUsage({
|
|
|
44
47
|
indexTokenCountMap,
|
|
45
48
|
toolExecution
|
|
46
49
|
);
|
|
50
|
+
context.fadingTier = fadingTier;
|
|
47
51
|
await context.tokenCalculationPromise;
|
|
48
52
|
return context.projectContextUsage(messages, {
|
|
49
53
|
runId,
|