@tanstack/ai-compaction 0.0.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +151 -5
- package/dist/esm/index.d.ts +148 -0
- package/dist/esm/index.js +343 -0
- package/dist/esm/index.js.map +1 -0
- package/dist/esm/index.test.d.ts +1 -0
- package/package.json +50 -10
- package/src/index.test.ts +543 -0
- package/src/index.ts +618 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Tanner Linsley
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,9 +1,155 @@
|
|
|
1
|
-
|
|
1
|
+
<div align="center">
|
|
2
|
+
<picture>
|
|
3
|
+
<source
|
|
4
|
+
media="(prefers-color-scheme: dark)"
|
|
5
|
+
srcset="https://tanstack.com/api/readme/ai.png?theme=dark"
|
|
6
|
+
/>
|
|
7
|
+
<source
|
|
8
|
+
media="(prefers-color-scheme: light)"
|
|
9
|
+
srcset="https://tanstack.com/api/readme/ai.png"
|
|
10
|
+
/>
|
|
11
|
+
<img
|
|
12
|
+
src="https://tanstack.com/api/readme/ai.png"
|
|
13
|
+
alt="TanStack AI"
|
|
14
|
+
width="900"
|
|
15
|
+
/>
|
|
16
|
+
</picture>
|
|
17
|
+
</div>
|
|
2
18
|
|
|
3
|
-
|
|
19
|
+
<br />
|
|
4
20
|
|
|
5
|
-
|
|
21
|
+
# @tanstack/ai-compaction
|
|
6
22
|
|
|
7
|
-
|
|
23
|
+
Context-window compaction as a `chat()` middleware. When the working message set
|
|
24
|
+
grows past `maxTokens`, `withCompaction` runs a pluggable **strategy** that
|
|
25
|
+
rewrites provider context. It runs before every model call, so compaction is
|
|
26
|
+
incremental and rolling. The canonical transcript and system prompt stay
|
|
27
|
+
unchanged.
|
|
8
28
|
|
|
9
|
-
|
|
29
|
+
```bash
|
|
30
|
+
npm install @tanstack/ai-compaction
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
The default strategy (`evictOldest`) drops the oldest messages and keeps the
|
|
36
|
+
recent ones.
|
|
37
|
+
|
|
38
|
+
```ts
|
|
39
|
+
import { chat } from '@tanstack/ai'
|
|
40
|
+
import { withCompaction } from '@tanstack/ai-compaction'
|
|
41
|
+
|
|
42
|
+
chat({
|
|
43
|
+
adapter,
|
|
44
|
+
messages,
|
|
45
|
+
middleware: [withCompaction({ maxTokens: 100_000 })],
|
|
46
|
+
})
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Strategies
|
|
50
|
+
|
|
51
|
+
Pass `strategy` to change how the history shrinks. Three are built in.
|
|
52
|
+
|
|
53
|
+
| Strategy | What it does | Cost |
|
|
54
|
+
| ----------------------- | ------------------------------------------------------- | ------------------- |
|
|
55
|
+
| `evictOldest` (default) | Drop the oldest messages, leave a marker | No extra model call |
|
|
56
|
+
| `summarizeOldest` | Replace the oldest messages with an LLM summary | One summarize call |
|
|
57
|
+
| `clearToolResults` | Stub the content of old tool results, keep the messages | No extra model call |
|
|
58
|
+
|
|
59
|
+
```ts
|
|
60
|
+
import {
|
|
61
|
+
withCompaction,
|
|
62
|
+
evictOldest,
|
|
63
|
+
summarizeOldest,
|
|
64
|
+
clearToolResults,
|
|
65
|
+
} from '@tanstack/ai-compaction'
|
|
66
|
+
|
|
67
|
+
// Tune how much recent history to keep.
|
|
68
|
+
withCompaction({
|
|
69
|
+
maxTokens: 100_000,
|
|
70
|
+
strategy: evictOldest({ keepRecentTokens: 40_000 }),
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
// Summarize instead of dropping. `summarize` gets the messages being removed.
|
|
74
|
+
withCompaction({
|
|
75
|
+
maxTokens: 100_000,
|
|
76
|
+
strategy: summarizeOldest({ summarize: (msgs) => summarizeToText(msgs) }),
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
// Best for agent loops: stub old tool output, keep the messages in place.
|
|
80
|
+
withCompaction({
|
|
81
|
+
maxTokens: 100_000,
|
|
82
|
+
strategy: clearToolResults({ keepRecentToolResults: 5 }),
|
|
83
|
+
})
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
### Combine them
|
|
87
|
+
|
|
88
|
+
`composeStrategies` runs strategies in order and escalates: it stops once the
|
|
89
|
+
result is back under `maxTokens`. Put the cheap one first.
|
|
90
|
+
|
|
91
|
+
```ts
|
|
92
|
+
import {
|
|
93
|
+
withCompaction,
|
|
94
|
+
composeStrategies,
|
|
95
|
+
clearToolResults,
|
|
96
|
+
evictOldest,
|
|
97
|
+
} from '@tanstack/ai-compaction'
|
|
98
|
+
|
|
99
|
+
// Clear old tool output first; only drop old messages if that isn't enough.
|
|
100
|
+
withCompaction({
|
|
101
|
+
maxTokens: 100_000,
|
|
102
|
+
strategy: composeStrategies(clearToolResults(), evictOldest()),
|
|
103
|
+
})
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### Write your own
|
|
107
|
+
|
|
108
|
+
A strategy gets the messages and the budget, and returns the rewritten messages
|
|
109
|
+
(or `null` to change nothing). It runs only when the estimate is over
|
|
110
|
+
`maxTokens`.
|
|
111
|
+
|
|
112
|
+
```ts
|
|
113
|
+
import type { CompactionStrategy } from '@tanstack/ai-compaction'
|
|
114
|
+
|
|
115
|
+
const keepLastOnly: CompactionStrategy = (messages) =>
|
|
116
|
+
messages.length <= 1 ? null : messages.slice(-1)
|
|
117
|
+
|
|
118
|
+
withCompaction({
|
|
119
|
+
maxTokens: 100_000,
|
|
120
|
+
strategy: keepLastOnly,
|
|
121
|
+
strategyKey: 'keep-last-v1',
|
|
122
|
+
})
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## Options
|
|
126
|
+
|
|
127
|
+
### `withCompaction`
|
|
128
|
+
|
|
129
|
+
| Option | Default | What it does |
|
|
130
|
+
| ---------------- | --------------- | ---------------------------------------------------------------------------- |
|
|
131
|
+
| `maxTokens` | (required) | Compact when estimated tokens exceed this. |
|
|
132
|
+
| `strategy` | `evictOldest()` | How to shrink the messages. |
|
|
133
|
+
| `estimateTokens` | chars / 4 | Per-message token estimate. Swap in a real tokenizer for accuracy. |
|
|
134
|
+
| `strategyKey` | built-in key | Stable checkpoint identity. Set it for custom strategies or estimators. |
|
|
135
|
+
| `onCompact` | — | Observe each compaction (`before`/`after`/`messagesBefore`/`messagesAfter`). |
|
|
136
|
+
|
|
137
|
+
### Strategy options
|
|
138
|
+
|
|
139
|
+
| Strategy | Options |
|
|
140
|
+
| ------------------ | ------------------------------------------------------------------------------- |
|
|
141
|
+
| `evictOldest` | `keepRecentTokens` (default `maxTokens / 2`), `marker` |
|
|
142
|
+
| `summarizeOldest` | `summarize` (required), `keepRecentTokens`, `summaryRole` (default `assistant`) |
|
|
143
|
+
| `clearToolResults` | `keepRecentToolResults` (default `3`), `stub` |
|
|
144
|
+
|
|
145
|
+
The token estimate is a rough `chars / 4` heuristic, good enough to trigger on,
|
|
146
|
+
not exact. Pass `estimateTokens` if you need provider-accurate counts.
|
|
147
|
+
|
|
148
|
+
When `withPersistence` provides a metadata store, compaction saves a validated
|
|
149
|
+
checkpoint automatically. The next request reuses the compacted prefix and adds
|
|
150
|
+
new canonical messages. Without metadata, compaction remains stateless.
|
|
151
|
+
|
|
152
|
+
TanStack AI DevTools has a Compaction tab with started, state, and ended
|
|
153
|
+
events, before/after counts, and dropped vs sent message previews. Those
|
|
154
|
+
stats ride the chat stream as `compaction:started`, `compaction:state`, and
|
|
155
|
+
`compaction:ended` CUSTOM events.
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
import { ChatMiddleware, ModelMessage } from '@tanstack/ai';
|
|
2
|
+
/** CUSTOM stream event: compaction is about to run. */
|
|
3
|
+
export declare const COMPACTION_STARTED_EVENT = "compaction:started";
|
|
4
|
+
/** CUSTOM stream event: compaction result (counts and previews). */
|
|
5
|
+
export declare const COMPACTION_STATE_EVENT = "compaction:state";
|
|
6
|
+
/** CUSTOM stream event: compaction finished. */
|
|
7
|
+
export declare const COMPACTION_ENDED_EVENT = "compaction:ended";
|
|
8
|
+
export type CompactionStreamEventName = typeof COMPACTION_STARTED_EVENT | typeof COMPACTION_STATE_EVENT | typeof COMPACTION_ENDED_EVENT;
|
|
9
|
+
/** One message in a `compaction:state` preview list. */
|
|
10
|
+
export interface CompactionMessagePreview {
|
|
11
|
+
role: string;
|
|
12
|
+
tokens: number;
|
|
13
|
+
text: string;
|
|
14
|
+
}
|
|
15
|
+
/** Payload of {@link COMPACTION_STARTED_EVENT}. */
|
|
16
|
+
export interface CompactionStartedEventValue {
|
|
17
|
+
before: number;
|
|
18
|
+
messagesBefore: number;
|
|
19
|
+
reusedCheckpoint: boolean;
|
|
20
|
+
maxTokens: number;
|
|
21
|
+
strategyKey?: string;
|
|
22
|
+
}
|
|
23
|
+
/** Payload of {@link COMPACTION_STATE_EVENT}. */
|
|
24
|
+
export interface CompactionStateEventValue {
|
|
25
|
+
before: number;
|
|
26
|
+
after: number;
|
|
27
|
+
messagesBefore: number;
|
|
28
|
+
messagesAfter: number;
|
|
29
|
+
reusedCheckpoint: boolean;
|
|
30
|
+
maxTokens: number;
|
|
31
|
+
strategyKey?: string;
|
|
32
|
+
/** Messages removed or rewritten. */
|
|
33
|
+
dropped?: Array<CompactionMessagePreview>;
|
|
34
|
+
/** Messages the model will see after compaction. */
|
|
35
|
+
result?: Array<CompactionMessagePreview>;
|
|
36
|
+
}
|
|
37
|
+
/** Payload of {@link COMPACTION_ENDED_EVENT}. */
|
|
38
|
+
export interface CompactionEndedEventValue {
|
|
39
|
+
after: number;
|
|
40
|
+
messagesAfter: number;
|
|
41
|
+
reusedCheckpoint: boolean;
|
|
42
|
+
maxTokens: number;
|
|
43
|
+
durationMs: number;
|
|
44
|
+
strategyKey?: string;
|
|
45
|
+
}
|
|
46
|
+
/** Rough token estimate for one message. Default: characters / 4. */
|
|
47
|
+
export declare function estimateMessageTokens(message: ModelMessage): number;
|
|
48
|
+
/** What a {@link CompactionStrategy} receives alongside the messages. */
|
|
49
|
+
export interface CompactionContext {
|
|
50
|
+
/** The `maxTokens` budget from `withCompaction`. */
|
|
51
|
+
maxTokens: number;
|
|
52
|
+
/** The shared token estimator (default {@link estimateMessageTokens}). */
|
|
53
|
+
estimate: (message: ModelMessage) => number;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Shrinks a message list. Called only when the estimate is over budget.
|
|
57
|
+
* Return the rewritten messages, or `null` to leave them unchanged.
|
|
58
|
+
*/
|
|
59
|
+
export type CompactionStrategy = (messages: ReadonlyArray<ModelMessage>, ctx: CompactionContext) => Array<ModelMessage> | null | Promise<Array<ModelMessage> | null>;
|
|
60
|
+
/** Reported to `onCompact` after each compaction event. */
|
|
61
|
+
export interface CompactionInfo {
|
|
62
|
+
/** Estimated tokens before compaction. */
|
|
63
|
+
before: number;
|
|
64
|
+
/** Estimated tokens after compaction. */
|
|
65
|
+
after: number;
|
|
66
|
+
/** Message count before compaction. */
|
|
67
|
+
messagesBefore: number;
|
|
68
|
+
/** Message count after compaction (unchanged for {@link clearToolResults}). */
|
|
69
|
+
messagesAfter: number;
|
|
70
|
+
}
|
|
71
|
+
export interface CompactionOptions {
|
|
72
|
+
/** Compact when estimated tokens across `messages` exceed this. */
|
|
73
|
+
maxTokens: number;
|
|
74
|
+
/** How to shrink the messages. Default: {@link evictOldest}. */
|
|
75
|
+
strategy?: CompactionStrategy;
|
|
76
|
+
/** Per-message token estimator. Default: {@link estimateMessageTokens}. */
|
|
77
|
+
estimateTokens?: (message: ModelMessage) => number;
|
|
78
|
+
/**
|
|
79
|
+
* Stable identity for persisted checkpoints. Set this for custom strategies
|
|
80
|
+
* or estimators, and change it when their output can change.
|
|
81
|
+
*/
|
|
82
|
+
strategyKey?: string;
|
|
83
|
+
/** Observe each compaction (logging, metrics). */
|
|
84
|
+
onCompact?: (info: CompactionInfo) => void;
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Drop the oldest messages and replace them with a short marker. Cheapest
|
|
88
|
+
* strategy — no extra model call. This is the default.
|
|
89
|
+
*/
|
|
90
|
+
export declare function evictOldest(options?: {
|
|
91
|
+
/** Tokens of recent messages to keep verbatim. Default `floor(maxTokens/2)`. */
|
|
92
|
+
keepRecentTokens?: number;
|
|
93
|
+
/** Build the marker that replaces the dropped head. */
|
|
94
|
+
marker?: (droppedCount: number) => string;
|
|
95
|
+
}): CompactionStrategy;
|
|
96
|
+
/**
|
|
97
|
+
* Drop the oldest messages and replace them with an LLM summary. Keeps the gist
|
|
98
|
+
* of old turns at the cost of one summarization call. Wire `summarize` to
|
|
99
|
+
* `summarize()` or any model call.
|
|
100
|
+
*/
|
|
101
|
+
export declare function summarizeOldest(options: {
|
|
102
|
+
summarize: (messages: Array<ModelMessage>) => Promise<string>;
|
|
103
|
+
/** Tokens of recent messages to keep verbatim. Default `floor(maxTokens/2)`. */
|
|
104
|
+
keepRecentTokens?: number;
|
|
105
|
+
/** Role of the injected summary message. Default `'assistant'`. */
|
|
106
|
+
summaryRole?: 'user' | 'assistant';
|
|
107
|
+
}): CompactionStrategy;
|
|
108
|
+
/**
|
|
109
|
+
* Replace the content of old tool-result messages with a stub, keeping every
|
|
110
|
+
* message and its tool-call pairing in place. Best for agent loops where tool
|
|
111
|
+
* output (file reads, command output) dominates the token count — it clears the
|
|
112
|
+
* bulk without disturbing the conversation shape. No extra model call.
|
|
113
|
+
*/
|
|
114
|
+
export declare function clearToolResults(options?: {
|
|
115
|
+
/** Number of most-recent tool results to keep verbatim. Default `3`. */
|
|
116
|
+
keepRecentToolResults?: number;
|
|
117
|
+
/** Text that replaces a cleared tool result. */
|
|
118
|
+
stub?: string;
|
|
119
|
+
}): CompactionStrategy;
|
|
120
|
+
/**
|
|
121
|
+
* Run several strategies in order, escalating: stop as soon as the running
|
|
122
|
+
* estimate is back under `maxTokens`. Put the cheap, targeted strategy first
|
|
123
|
+
* (for example {@link clearToolResults}) and a broad fallback last (for example
|
|
124
|
+
* {@link evictOldest}) — the fallback only runs when clearing was not enough.
|
|
125
|
+
* A strategy that returns `null` (no change) is skipped and the next one runs.
|
|
126
|
+
*
|
|
127
|
+
* @example
|
|
128
|
+
* ```ts
|
|
129
|
+
* withCompaction({
|
|
130
|
+
* maxTokens: 100_000,
|
|
131
|
+
* strategy: composeStrategies(clearToolResults(), evictOldest()),
|
|
132
|
+
* })
|
|
133
|
+
* ```
|
|
134
|
+
*/
|
|
135
|
+
export declare function composeStrategies(...strategies: Array<CompactionStrategy>): CompactionStrategy;
|
|
136
|
+
/**
|
|
137
|
+
* Context-compaction middleware. Add to `chat({ middleware: [...] })`.
|
|
138
|
+
*
|
|
139
|
+
* @example
|
|
140
|
+
* ```ts
|
|
141
|
+
* chat({
|
|
142
|
+
* adapter,
|
|
143
|
+
* messages,
|
|
144
|
+
* middleware: [withCompaction({ maxTokens: 100_000 })], // evictOldest by default
|
|
145
|
+
* })
|
|
146
|
+
* ```
|
|
147
|
+
*/
|
|
148
|
+
export declare function withCompaction(options: CompactionOptions): ChatMiddleware;
|
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
import { MetadataCapability, getMetadata } from "@tanstack/ai";
|
|
2
|
+
//#region src/index.ts
|
|
3
|
+
/**
|
|
4
|
+
* `@tanstack/ai-compaction` — context-window compaction as a `chat()`
|
|
5
|
+
* middleware. `withCompaction({ maxTokens, strategy })` runs before each model
|
|
6
|
+
* call: when the working message set grows past `maxTokens`, the chosen
|
|
7
|
+
* `CompactionStrategy` rewrites the messages. Because it runs every call,
|
|
8
|
+
* compaction is incremental and rolling.
|
|
9
|
+
*
|
|
10
|
+
* Strategies are pluggable, mirroring `AgentLoopStrategy`. Three are built in:
|
|
11
|
+
* {@link evictOldest}, {@link summarizeOldest}, and {@link clearToolResults}.
|
|
12
|
+
* Write your own by passing any {@link CompactionStrategy}.
|
|
13
|
+
*
|
|
14
|
+
* The system prompt is never touched — `chat()` keeps it separate from
|
|
15
|
+
* `messages`.
|
|
16
|
+
*/
|
|
17
|
+
/** CUSTOM stream event: compaction is about to run. */
|
|
18
|
+
var COMPACTION_STARTED_EVENT = "compaction:started";
|
|
19
|
+
/** CUSTOM stream event: compaction result (counts and previews). */
|
|
20
|
+
var COMPACTION_STATE_EVENT = "compaction:state";
|
|
21
|
+
/** CUSTOM stream event: compaction finished. */
|
|
22
|
+
var COMPACTION_ENDED_EVENT = "compaction:ended";
|
|
23
|
+
var PREVIEW_CHARS = 4e3;
|
|
24
|
+
var MAX_PREVIEWS = 24;
|
|
25
|
+
function emitCompactionStarted(ctx, value) {
|
|
26
|
+
ctx.emitCustomEvent(COMPACTION_STARTED_EVENT, value);
|
|
27
|
+
}
|
|
28
|
+
function emitCompactionState(ctx, value) {
|
|
29
|
+
ctx.emitCustomEvent(COMPACTION_STATE_EVENT, value);
|
|
30
|
+
}
|
|
31
|
+
function emitCompactionEnded(ctx, value) {
|
|
32
|
+
ctx.emitCustomEvent(COMPACTION_ENDED_EVENT, value);
|
|
33
|
+
}
|
|
34
|
+
var strategyKeys = /* @__PURE__ */ new WeakMap();
|
|
35
|
+
var CHECKPOINT_NAMESPACE = "@tanstack/ai-compaction";
|
|
36
|
+
function identifyStrategy(strategy, key) {
|
|
37
|
+
if (key) strategyKeys.set(strategy, key);
|
|
38
|
+
return strategy;
|
|
39
|
+
}
|
|
40
|
+
async function hashMessages(messages) {
|
|
41
|
+
const bytes = new TextEncoder().encode(JSON.stringify(messages));
|
|
42
|
+
const digest = await globalThis.crypto.subtle.digest("SHA-256", bytes);
|
|
43
|
+
return Array.from(new Uint8Array(digest), (byte) => byte.toString(16).padStart(2, "0")).join("");
|
|
44
|
+
}
|
|
45
|
+
function isModelMessage(value) {
|
|
46
|
+
return typeof value === "object" && value !== null && "role" in value && (value.role === "user" || value.role === "assistant" || value.role === "tool") && "content" in value;
|
|
47
|
+
}
|
|
48
|
+
function isCompactionCheckpoint(value) {
|
|
49
|
+
return typeof value === "object" && value !== null && "schemaVersion" in value && value.schemaVersion === 1 && "sourceMessageCount" in value && typeof value.sourceMessageCount === "number" && Number.isInteger(value.sourceMessageCount) && value.sourceMessageCount >= 0 && "sourceHash" in value && typeof value.sourceHash === "string" && "strategyKey" in value && typeof value.strategyKey === "string" && "compactedMessages" in value && Array.isArray(value.compactedMessages) && value.compactedMessages.every(isModelMessage);
|
|
50
|
+
}
|
|
51
|
+
function messagePreviewText(message) {
|
|
52
|
+
if (typeof message.content === "string") return message.content;
|
|
53
|
+
return JSON.stringify(message.content ?? "");
|
|
54
|
+
}
|
|
55
|
+
function toMessagePreview(message, estimate) {
|
|
56
|
+
const text = messagePreviewText(message);
|
|
57
|
+
return {
|
|
58
|
+
role: message.role,
|
|
59
|
+
tokens: estimate(message),
|
|
60
|
+
text: text.length > PREVIEW_CHARS ? `${text.slice(0, PREVIEW_CHARS)}…` : text
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
function previewList(messages, estimate) {
|
|
64
|
+
const mapped = messages.map((message) => toMessagePreview(message, estimate));
|
|
65
|
+
if (mapped.length <= MAX_PREVIEWS) return mapped;
|
|
66
|
+
return mapped.slice(0, MAX_PREVIEWS);
|
|
67
|
+
}
|
|
68
|
+
function droppedMessages(before, after) {
|
|
69
|
+
const afterKeys = new Set(after.map((message) => JSON.stringify(message)));
|
|
70
|
+
return before.filter((message) => !afterKeys.has(JSON.stringify(message)));
|
|
71
|
+
}
|
|
72
|
+
function compactionStateValue(args) {
|
|
73
|
+
const value = {
|
|
74
|
+
before: args.before,
|
|
75
|
+
after: args.after,
|
|
76
|
+
messagesBefore: args.messagesBefore,
|
|
77
|
+
messagesAfter: args.messagesAfter,
|
|
78
|
+
reusedCheckpoint: args.reusedCheckpoint,
|
|
79
|
+
maxTokens: args.maxTokens,
|
|
80
|
+
...args.strategyKey ? { strategyKey: args.strategyKey } : {}
|
|
81
|
+
};
|
|
82
|
+
if (args.afterMessages) value.result = previewList(args.afterMessages, args.estimate);
|
|
83
|
+
if (args.beforeMessages && args.afterMessages) value.dropped = previewList(droppedMessages(args.beforeMessages, args.afterMessages), args.estimate);
|
|
84
|
+
return value;
|
|
85
|
+
}
|
|
86
|
+
/** Rough token estimate for one message. Default: characters / 4. */
|
|
87
|
+
function estimateMessageTokens(message) {
|
|
88
|
+
let text = messagePreviewText(message);
|
|
89
|
+
if (message.toolCalls?.length) text += JSON.stringify(message.toolCalls);
|
|
90
|
+
return Math.ceil(text.length / 4);
|
|
91
|
+
}
|
|
92
|
+
var sum = (messages, estimate) => messages.reduce((total, m) => total + estimate(m), 0);
|
|
93
|
+
/**
|
|
94
|
+
* Find the split point that keeps the most recent messages up to
|
|
95
|
+
* `keepRecentTokens`, then moves the cut forward past any leading tool result
|
|
96
|
+
* so the kept tail never starts with an orphan (its tool call would be dropped).
|
|
97
|
+
* Returns the index where the tail begins (head is `messages[0..cut)`).
|
|
98
|
+
*/
|
|
99
|
+
function splitAtRecent(messages, estimate, keepRecentTokens) {
|
|
100
|
+
let kept = 0;
|
|
101
|
+
let cut = messages.length;
|
|
102
|
+
while (cut > 0) {
|
|
103
|
+
const prev = messages[cut - 1];
|
|
104
|
+
if (!prev) break;
|
|
105
|
+
const size = estimate(prev);
|
|
106
|
+
if (kept + size > keepRecentTokens) break;
|
|
107
|
+
kept += size;
|
|
108
|
+
cut--;
|
|
109
|
+
}
|
|
110
|
+
if (cut >= messages.length) cut = messages.length - 1;
|
|
111
|
+
while (cut < messages.length && messages[cut]?.role === "tool") cut++;
|
|
112
|
+
if (cut >= messages.length) {
|
|
113
|
+
cut = messages.length;
|
|
114
|
+
while (cut > 0 && messages[cut - 1]?.role === "tool") cut--;
|
|
115
|
+
if (cut > 0) cut--;
|
|
116
|
+
}
|
|
117
|
+
return cut;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Drop the oldest messages and replace them with a short marker. Cheapest
|
|
121
|
+
* strategy — no extra model call. This is the default.
|
|
122
|
+
*/
|
|
123
|
+
function evictOldest(options = {}) {
|
|
124
|
+
const strategy = (messages, ctx) => {
|
|
125
|
+
const keep = options.keepRecentTokens ?? Math.floor(ctx.maxTokens / 2);
|
|
126
|
+
const cut = splitAtRecent(messages, ctx.estimate, keep);
|
|
127
|
+
if (cut <= 0) return null;
|
|
128
|
+
return [{
|
|
129
|
+
role: "user",
|
|
130
|
+
content: options.marker?.(cut) ?? `[${cut} earlier message(s) omitted to save context.]`
|
|
131
|
+
}, ...messages.slice(cut)];
|
|
132
|
+
};
|
|
133
|
+
return identifyStrategy(strategy, options.marker ? void 0 : `evict-oldest:${options.keepRecentTokens ?? "half"}`);
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Drop the oldest messages and replace them with an LLM summary. Keeps the gist
|
|
137
|
+
* of old turns at the cost of one summarization call. Wire `summarize` to
|
|
138
|
+
* `summarize()` or any model call.
|
|
139
|
+
*/
|
|
140
|
+
function summarizeOldest(options) {
|
|
141
|
+
const strategy = async (messages, ctx) => {
|
|
142
|
+
const keep = options.keepRecentTokens ?? Math.floor(ctx.maxTokens / 2);
|
|
143
|
+
const cut = splitAtRecent(messages, ctx.estimate, keep);
|
|
144
|
+
if (cut <= 0) return null;
|
|
145
|
+
const summary = await options.summarize(messages.slice(0, cut));
|
|
146
|
+
return [{
|
|
147
|
+
role: options.summaryRole ?? "assistant",
|
|
148
|
+
content: `<untrusted-conversation-summary>\n${summary}\n</untrusted-conversation-summary>`
|
|
149
|
+
}, ...messages.slice(cut)];
|
|
150
|
+
};
|
|
151
|
+
return identifyStrategy(strategy, `summarize-oldest:${options.keepRecentTokens ?? "half"}:${options.summaryRole ?? "assistant"}`);
|
|
152
|
+
}
|
|
153
|
+
/**
|
|
154
|
+
* Replace the content of old tool-result messages with a stub, keeping every
|
|
155
|
+
* message and its tool-call pairing in place. Best for agent loops where tool
|
|
156
|
+
* output (file reads, command output) dominates the token count — it clears the
|
|
157
|
+
* bulk without disturbing the conversation shape. No extra model call.
|
|
158
|
+
*/
|
|
159
|
+
function clearToolResults(options = {}) {
|
|
160
|
+
const keepN = options.keepRecentToolResults ?? 3;
|
|
161
|
+
const stub = options.stub ?? "[tool output cleared to save context]";
|
|
162
|
+
const strategy = (messages) => {
|
|
163
|
+
const toolIndexes = [];
|
|
164
|
+
messages.forEach((m, i) => {
|
|
165
|
+
if (m.role === "tool") toolIndexes.push(i);
|
|
166
|
+
});
|
|
167
|
+
if (toolIndexes.length <= keepN) return null;
|
|
168
|
+
const clearBefore = toolIndexes[toolIndexes.length - keepN] ?? 0;
|
|
169
|
+
let changed = false;
|
|
170
|
+
const next = messages.map((m, i) => {
|
|
171
|
+
if (m.role === "tool" && i < clearBefore && m.content !== stub) {
|
|
172
|
+
changed = true;
|
|
173
|
+
return {
|
|
174
|
+
...m,
|
|
175
|
+
content: stub
|
|
176
|
+
};
|
|
177
|
+
}
|
|
178
|
+
return m;
|
|
179
|
+
});
|
|
180
|
+
return changed ? next : null;
|
|
181
|
+
};
|
|
182
|
+
return identifyStrategy(strategy, `clear-tool-results:${keepN}:${stub}`);
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Run several strategies in order, escalating: stop as soon as the running
|
|
186
|
+
* estimate is back under `maxTokens`. Put the cheap, targeted strategy first
|
|
187
|
+
* (for example {@link clearToolResults}) and a broad fallback last (for example
|
|
188
|
+
* {@link evictOldest}) — the fallback only runs when clearing was not enough.
|
|
189
|
+
* A strategy that returns `null` (no change) is skipped and the next one runs.
|
|
190
|
+
*
|
|
191
|
+
* @example
|
|
192
|
+
* ```ts
|
|
193
|
+
* withCompaction({
|
|
194
|
+
* maxTokens: 100_000,
|
|
195
|
+
* strategy: composeStrategies(clearToolResults(), evictOldest()),
|
|
196
|
+
* })
|
|
197
|
+
* ```
|
|
198
|
+
*/
|
|
199
|
+
function composeStrategies(...strategies) {
|
|
200
|
+
const strategy = async (messages, ctx) => {
|
|
201
|
+
let current = messages;
|
|
202
|
+
let result = null;
|
|
203
|
+
for (const itemStrategy of strategies) {
|
|
204
|
+
if (sum(current, ctx.estimate) <= ctx.maxTokens) break;
|
|
205
|
+
const out = await itemStrategy(current, ctx);
|
|
206
|
+
if (out) {
|
|
207
|
+
current = out;
|
|
208
|
+
result = out;
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
return result;
|
|
212
|
+
};
|
|
213
|
+
const keys = strategies.map((item) => strategyKeys.get(item));
|
|
214
|
+
return identifyStrategy(strategy, keys.every((key) => key !== void 0) ? keys.join("|") : void 0);
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* Context-compaction middleware. Add to `chat({ middleware: [...] })`.
|
|
218
|
+
*
|
|
219
|
+
* @example
|
|
220
|
+
* ```ts
|
|
221
|
+
* chat({
|
|
222
|
+
* adapter,
|
|
223
|
+
* messages,
|
|
224
|
+
* middleware: [withCompaction({ maxTokens: 100_000 })], // evictOldest by default
|
|
225
|
+
* })
|
|
226
|
+
* ```
|
|
227
|
+
*/
|
|
228
|
+
function withCompaction(options) {
|
|
229
|
+
const estimate = options.estimateTokens ?? estimateMessageTokens;
|
|
230
|
+
const strategy = options.strategy ?? evictOldest();
|
|
231
|
+
const strategyKey = options.strategyKey ?? (options.estimateTokens ? void 0 : strategyKeys.get(strategy));
|
|
232
|
+
const checkpointStrategyKey = strategyKey ? `${strategyKey}:maxTokens=${options.maxTokens}` : void 0;
|
|
233
|
+
return {
|
|
234
|
+
name: "compaction",
|
|
235
|
+
optionalRequires: [MetadataCapability],
|
|
236
|
+
async onConfig(ctx, config) {
|
|
237
|
+
if (ctx.phase === "init") return;
|
|
238
|
+
const startedAt = Date.now();
|
|
239
|
+
const { messages } = config;
|
|
240
|
+
const inputMessages = config.providerMessages ?? messages;
|
|
241
|
+
const metadata = getMetadata(ctx, { optional: true });
|
|
242
|
+
let workingMessages = inputMessages;
|
|
243
|
+
let reusedCheckpoint = false;
|
|
244
|
+
if (metadata && checkpointStrategyKey && inputMessages === messages) {
|
|
245
|
+
const stored = await metadata.get(CHECKPOINT_NAMESPACE, ctx.threadId);
|
|
246
|
+
if (isCompactionCheckpoint(stored) && stored.strategyKey === checkpointStrategyKey && stored.sourceMessageCount <= messages.length && stored.sourceHash === await hashMessages(messages.slice(0, stored.sourceMessageCount))) {
|
|
247
|
+
workingMessages = [...stored.compactedMessages, ...messages.slice(stored.sourceMessageCount)];
|
|
248
|
+
reusedCheckpoint = true;
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
const before = sum(workingMessages, estimate);
|
|
252
|
+
const startedValue = {
|
|
253
|
+
before,
|
|
254
|
+
messagesBefore: workingMessages.length,
|
|
255
|
+
reusedCheckpoint,
|
|
256
|
+
maxTokens: options.maxTokens,
|
|
257
|
+
...checkpointStrategyKey ? { strategyKey: checkpointStrategyKey } : {}
|
|
258
|
+
};
|
|
259
|
+
if (before <= options.maxTokens) {
|
|
260
|
+
if (reusedCheckpoint) {
|
|
261
|
+
emitCompactionStarted(ctx, startedValue);
|
|
262
|
+
emitCompactionState(ctx, compactionStateValue({
|
|
263
|
+
before,
|
|
264
|
+
after: before,
|
|
265
|
+
messagesBefore: workingMessages.length,
|
|
266
|
+
messagesAfter: workingMessages.length,
|
|
267
|
+
reusedCheckpoint: true,
|
|
268
|
+
maxTokens: options.maxTokens,
|
|
269
|
+
strategyKey: checkpointStrategyKey,
|
|
270
|
+
afterMessages: workingMessages,
|
|
271
|
+
estimate
|
|
272
|
+
}));
|
|
273
|
+
emitCompactionEnded(ctx, {
|
|
274
|
+
after: before,
|
|
275
|
+
messagesAfter: workingMessages.length,
|
|
276
|
+
reusedCheckpoint: true,
|
|
277
|
+
maxTokens: options.maxTokens,
|
|
278
|
+
durationMs: Date.now() - startedAt,
|
|
279
|
+
...checkpointStrategyKey ? { strategyKey: checkpointStrategyKey } : {}
|
|
280
|
+
});
|
|
281
|
+
return { providerMessages: workingMessages };
|
|
282
|
+
}
|
|
283
|
+
return;
|
|
284
|
+
}
|
|
285
|
+
emitCompactionStarted(ctx, startedValue);
|
|
286
|
+
const next = await strategy(workingMessages, {
|
|
287
|
+
maxTokens: options.maxTokens,
|
|
288
|
+
estimate
|
|
289
|
+
});
|
|
290
|
+
if (!next || next === workingMessages) {
|
|
291
|
+
emitCompactionEnded(ctx, {
|
|
292
|
+
after: before,
|
|
293
|
+
messagesAfter: workingMessages.length,
|
|
294
|
+
reusedCheckpoint,
|
|
295
|
+
maxTokens: options.maxTokens,
|
|
296
|
+
durationMs: Date.now() - startedAt,
|
|
297
|
+
...checkpointStrategyKey ? { strategyKey: checkpointStrategyKey } : {}
|
|
298
|
+
});
|
|
299
|
+
if (reusedCheckpoint) return { providerMessages: workingMessages };
|
|
300
|
+
return;
|
|
301
|
+
}
|
|
302
|
+
const info = {
|
|
303
|
+
before,
|
|
304
|
+
after: sum(next, estimate),
|
|
305
|
+
messagesBefore: workingMessages.length,
|
|
306
|
+
messagesAfter: next.length
|
|
307
|
+
};
|
|
308
|
+
options.onCompact?.(info);
|
|
309
|
+
emitCompactionState(ctx, compactionStateValue({
|
|
310
|
+
...info,
|
|
311
|
+
reusedCheckpoint,
|
|
312
|
+
maxTokens: options.maxTokens,
|
|
313
|
+
strategyKey: checkpointStrategyKey,
|
|
314
|
+
beforeMessages: workingMessages,
|
|
315
|
+
afterMessages: next,
|
|
316
|
+
estimate
|
|
317
|
+
}));
|
|
318
|
+
emitCompactionEnded(ctx, {
|
|
319
|
+
after: info.after,
|
|
320
|
+
messagesAfter: info.messagesAfter,
|
|
321
|
+
reusedCheckpoint,
|
|
322
|
+
maxTokens: options.maxTokens,
|
|
323
|
+
durationMs: Date.now() - startedAt,
|
|
324
|
+
...checkpointStrategyKey ? { strategyKey: checkpointStrategyKey } : {}
|
|
325
|
+
});
|
|
326
|
+
if (metadata && checkpointStrategyKey && inputMessages === messages) {
|
|
327
|
+
const checkpoint = {
|
|
328
|
+
schemaVersion: 1,
|
|
329
|
+
sourceMessageCount: messages.length,
|
|
330
|
+
sourceHash: await hashMessages(messages),
|
|
331
|
+
strategyKey: checkpointStrategyKey,
|
|
332
|
+
compactedMessages: next
|
|
333
|
+
};
|
|
334
|
+
if (!ctx.signal?.aborted) await metadata.set(CHECKPOINT_NAMESPACE, ctx.threadId, checkpoint);
|
|
335
|
+
}
|
|
336
|
+
return { providerMessages: next };
|
|
337
|
+
}
|
|
338
|
+
};
|
|
339
|
+
}
|
|
340
|
+
//#endregion
|
|
341
|
+
export { COMPACTION_ENDED_EVENT, COMPACTION_STARTED_EVENT, COMPACTION_STATE_EVENT, clearToolResults, composeStrategies, estimateMessageTokens, evictOldest, summarizeOldest, withCompaction };
|
|
342
|
+
|
|
343
|
+
//# sourceMappingURL=index.js.map
|